From ec0096a591a2e728a469ab7d8e9353818e5cd3a4 Mon Sep 17 00:00:00 2001 From: Oliphant <131810276+Oliphant714@users.noreply.github.com> Date: Thu, 23 Apr 2026 15:15:28 -0600 Subject: [PATCH 1/7] analyze-repo.md --- .../__pycache__/build-index.cpython-313.pyc | Bin 0 -> 15151 bytes agents/__pycache__/hash-files.cpython-313.pyc | Bin 0 -> 3886 bytes agents/analyze-repo.md | 173 ++++++++++++ agents/build-index.py | 265 ++++++++++++++++++ agents/hash-files.py | 70 +++++ agents/list-files.sh | 16 ++ 6 files changed, 524 insertions(+) create mode 100644 agents/__pycache__/build-index.cpython-313.pyc create mode 100644 agents/__pycache__/hash-files.cpython-313.pyc create mode 100644 agents/analyze-repo.md create mode 100644 agents/build-index.py create mode 100644 agents/hash-files.py create mode 100644 agents/list-files.sh diff --git a/agents/__pycache__/build-index.cpython-313.pyc b/agents/__pycache__/build-index.cpython-313.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ee70f1f5306147df142b8b3bfee0a7b6406e8170 GIT binary patch literal 15151 zcmc(GZE#!HmDqj2!#BV$@S79~ehZ{X$`bWKn)((+Ni->$kB}&n5)6VM2@?e9`#^m# zvdtvjH5I27RoN>l$*ky|OwDZCiMH#`syo|VDakZ$`=fY0we zJR$E9flwlFg06=1N$1Ok$p|n1Yf?#eA+@Sr>)e=>KS@$ zr|s0vU}KNtw1YZMXVEN{HuYqm&Y?LBHuvP7&ZBuQvXclF!RqA%n_!37!Qw0-8{ixk z=L&fMJ6W7B6aZW(6hT}plt5f6ltEn1;%!0&z?Cd^vA9a8hMXFq7UDYiS1&XO+gtc9 zjyi{p{oB>)NPnw|;NHV2X`!)AwvP)NsbReUZR)q| z+b#|7d2R02uwVpR(rVMN_Wb{8bIYFX)q1w?IsM#{(D0r&8!u{D(B^Yi zy)zSU%Qg>aSkLF7gBl*T%56#S(0V-KaEqB14BLUHj|fNGxncs)u~H%&-=kAFd{2t; zX`%44&{axa<7tu5FPvadF+kl6DiH+XB!fzY0pS#b$^?&annC5lIiZI^+l2GNOAM+I z28CV*RSFk`J_fmjSA;VRsv52q&UOqE;&Tu(TvJVktE!3G7j+B)Zd~sknr%Ni6AX>I zXka=b2BQ&r&E*UGLf6Eg=n95M16Kp0Yb-)t(TPC1z&9QUM_WF{!MQov<_m`-QC~C| z35$JhooqT5nF&V&RJM%<#(XoO=xER%1w6~BFY5D$e4+?QW=I60!Kr|3#&|Xo4gd;& z))$?CulW?z_l<-Cv-+0lYqD|rT67{3h6Ejvc{($4ZPw62eUn)HnkZXe^Ii6dej1#P z&hn6P6|$p{9lbUk(6XZ}dpsiR#v@2bL_%T|2w`oR$9PuP;+Hu;d?o1iOT z#1jBRt|%M*;)IW4Vro=2PxVURp>k(}#GCC6qY(*FGWTS$kkcD1sgnZn&`xp2$3+%r;_wg2?-0KL3~0uxNO*ZR&`zjYA#7+lMZ> zFE$NB*BfC#Td>A#2iDryv&`9q$y5l6Q99#~Ue_9Q&mKxm-1N+dI%_GUKzL^AdX)Nt z(RDMof$r>%4Ftx*ne(#|yDNx7~p-K*=^^m7~S`$}X|1EB$bgpNLsp4&`q zAi<~|BM_8?W?__#Q4U7A80AAW03yjaMkBL?8p1kzqnPu=I2SNA-0? z&<*o~z5z-E9v}mQFwdw#CJiz($U+U{R<})d^!mcVv49xu3Ph&r+ z7ht>B`}&M8lq$*gddGrPjD~{YKsW-$kdYzOA)~?ORLS_C8=I(2a^}LIG`TZl;k)ROVV{v-{JE z81qn{rqrm+gHD;AwKO+JW}?lbK{{J`JP2&gls4*U(8{1Sqk&OXbI;~AH&6MlHjCG$ zMk1l+k!!GO!G)Nf%+?E&!D*M0?*iVTKBg*MzLCh~04RIV{>Wc$J*|L&(0v%~hX@(U zN5`jqR1C;jN9g#>6qo{MF-c{U*E<^VBX{{mN443l91MU#G zW>`@{hjEzmaQ&IstP0VrEA`0Z`!`TSKa`RXgm^gB?#jp`YF_JC#{{c5tn=uc$ zMqiP^@BPM9XEHw3#Djc}{|xPXnD_AgDCIVUeiU^Zg37T@a?T*?T8W5lwi6QZ^gB18 zLQ=(6$3F^l*INSPHarizzu=3M1fyV5?dP+)-WLh65f`)`hF-WxElUj*sr5YC9M+O5 zlHQL)sS`NKoFA0*8$z4jM0_AIT%?lp{*aI{azI~6#yD&X+Z-cqbDwNfq}!|`6cnS) zW5G~B1RGA~M||SMfLjk{+C;!VDQEe8)6p3k@WKY2o&gyhg{>uX!!pnf^>%-!`|hYm@z7nM>hK-Q&Yb1s3@CK zkZhy@5iE9@M^zvqKeie=ON=tsw6=ZB<{;RWNWksRT=xt}>6ITD8lrNe`d55E$3 zPJVp&l?Qy{(u8zrB5~=pc>eB%<9ANqI(_@4ME-UufBSO&viKx_w{-Xw32^O`Pi>^w zwPq*wlEr*Q0Q_dnL<;J^f8|4KVrQ?kvp44Kdqy}%iJViss)z4C{=IWjL@D?vx9C`* z?#F!bu}yWuDMIf&PZ5MU@pgzN=T@g8YKu|YNy8-gYlU~^rC4Gmkn zUlSTYB^5T^UZt?Tbo~YuC;d%tip^A?N-M1Q==&|{S}G*yRI#LO=0a#`b6ovKTpw=y zaTIv{o2WsLhVLT1Ai!{I525YFLRv_In&kUrt`Uw*(}D0P*i3L(2#mUU+7GSD z2DUk99l~aniij(SoUbS+v|cDi5%`JMAp&kMDw*p_=H%b8-m)(4{-f-9KAG*jdHu%q zn{VED^XK;bq_ZgDtdX2GOUL6*cfz?#a_)*d+vd$ln=@gnkZcu;R}!`c$=0yEKhbnh zYC8Dftwhrqsp$+>Hal+GZ`c{fp z<;yvuR^bRRyx^Zo+eHSe`ZXSbk*0z=hBh6F{vAZ)R+xh^5Z{%+TQ1U9sR(qpp^pM< z5?Mz_{3;3j;%)iH-^`Jv`VX4kZ(5#wT-~AA!e|f! zvzxGfC^$lKh%z4mHkSDyx>)of0F5)#=uMF=6M?Iv!Erb@xD5(F8q^2D0~IrK=XLgH(i)uOSMuI`nD9k$>~jw=TsU+d!n->~qJH7RSw!8zl>E->zId zu~Ja-xU}QJu_pzG=ev^joOh2c8g8GK>=pCJllHuuy*GLn>z>$MPYtB-2=}?3SW8yT z#GaGLt&(!9;<+_3TTP6w`P>2p|Ge^dB1()u-FKuz|ER-y)NEQOQ?bj8$3aZuF#u

mW~N#>PO_7($UN zup|X99qH@#9O>@z9=+i08XV~A6HcA!ll9{PH0B+ugz@50j=F(>h`mlp8aZox22294 zdWNI86n}t7Z4kj~4+@~;74$_ch6Y~TUhz=L{b<)WRaO)#LPf;w^9Rs*`dpc^x6 zT3%HMGf*~tZU~o+g|s+eC09XpX(C0%0ZN+q7~C^B?lJF4)mN#4+Bl~O2@fs@*9WUG z4AmLmve}}qVzL?{w?&cgS?lH;tQ|PA&|pBMpn!;MLHVz&25u9!$Jn~iMo5a-AKWP{ zq|6|8$n}Eh#183n4p6>@xUWJ4ilL@%uKSachIvOazZguAc}p_8IFap=vR#SnYAL%K z&#zSMSiTsqcu_K!0jfFsX4Z|Yci&jL_@Mry!FbWRC+72@0ZO;8 zf(CHT^?gzJoV3$*Uv-!4MCmlaK^bLtqxkb z3}vA2H*cs6Nf6(9UTeDD$OW)65X9s#cV1tBR;(L}t*sJ1WYm=_y3Nzk3i_89xo zIkO?CJQ+3(79q2Q%|qI0#sew}^E7$}>s~Z-2okz3I4U z!3Q(8IhMzOeK});((;WIU=}UVTPy4B7X9e`rLqsZxn7tUG~{-YuoaYiGTM$F_5J zQJ*>=gG6K8{8t_uoLxb7u(PW^Y!AcjYl~sTh zI{N(SJ`#eOjafEF0f0+r&}C_@sV>P^$q{rd1`xJ-#e#AhWGj=Ie!SeFQ4}}Kaj2L$yD=GrNUm4J>R2Q+8lYZr2F|AR+7}I` zJAf;0g9s#TVNtTEc%`8H_Jx(gZMTCf`DM3HeP%Nr;fxCLy@2-V|As)ni4bZu3X@Oh zA7J!{5XpqT4WCa@DS|3Nw2c@!c)JC6QC@8jPKu1nw7+6!!EsrP$|mq?foc)K8w;*% z=89v7eR>J8(Y$1a7}G0?h6zKSY=CM|iAQIwYyh)v3KDo{MtxXpn~K1J8xQD_u-if3 z#5_}Kc4PypvTI5wOv=oG8-SgXz+=s3Qr1rf;0kSI#y=SVcL5s&a|z2v#Z5u~C6Y8k z6R?~j$Sac4c9}Xaf3wA*OMv`G@DpP&wP2m>*!Pqe+l|kNvDq}&y_Q4rO6CnK+0KP7 zJd?+>D?!^iix%~_+ZUR?&o2#pF#P`T{hIr`Vs(4tu6UM=Z_@|%5M$NpIE7>U%I>;k{Vu&*K|I3U8*@e ze`29ivb$Dl8eEcjXJqf)812MP~J_9`$;7haU?l}m;Xvfj^%+gtCy zAlcjCiv?JFso;ak_bU@sd+#5O+xIL#UHN^Oa-#)Vd zE_46<$Ih(o{W*R;EV3GKt3aB2_6^X>nfAsIq?MpbM>r6oWx0y_A1q4T`Wu44p@tuY z`xYqIOz||5N7wN#XfdGN3vxK^-`mu=uxSge+CegG@E9_6Y;PYSgM9xcUO{F98!O1> z7Kd_o(5`!)*EDPzDo{t^fm;N@*pB|w@sxhJ;W0o83|FQte(16GYaUIt7R-o6n@79H zzOLt>V?FkM%??P{5e!-#$KU}JQ~hP>QWb)^-elAsv&Xi@Dw$d|n++^*kY}jZdO>Wn z?`J7}hEkb@A_{IRP>msAU@Hn_$(V;3*w9?_Q4t`P)x!?c-mLl{#L^3}K zLX%16b%##S=L9qq_v|nLHPAZV2f)gK&Bsl}Gl`|$Fk9kH_mG0%{6c33)ck!eoQ)8B=1xT#`F z2wdE{mYs^Cu%HKnJ@ycB1fV2RK7>xS9l83;`@*`=0`)(IPJnzUI;jILy2^Fyq){<8 zIf&5ar)8 z6om-W%*-FOP6?;xID5my(!YeN_!B#TA3y=`I`CJc@ma0#U>m_IX(#!{MzW?TZ8h3WgoBuAm2%V4l#rHbyc1Zbk z^A_;OTMClSas~Om>7hC1tdBWQC7c72b0F?KH-CK1NUEEb+wQ;qxT;;UZ=3J=q_W|b zZNZQ%-geJ+*9Pv-V)ugK?~2NQo?o7xmCIqRn5hGQXc=e1j}B?=p)!iHs2ym04Z^9xC}=&`w6gYSfHg%X8ssnEUL z5HH;I*t}aSN~68&U#z>=c(?I(BvH6SD%`PrK3=%zv3akC`ovrg&L%Kb@wy>lE0b(x zaoe^Szl~`He%z)c>a8zkKV>35X zH5tr<>taqYv4*!4w85k6JgX@h< z%G?!->jvua`|$Y`DT2Xa#DfVr@)V z+)pfS#TWm>?MKS>kIJn_8clv^MhiB8*xC<6oOuLLjk<8AOPFkC_^Ye3nG;x}g14r~ zXB&cQ3j1o-xt=%0{VMI8F`=J7zzOfb-xnY*A`trN%8vO<4LBdRfnUxry7+WA3j|N1gsct}C5V zA;H|Dhfx7Dpo#q@qq1z(Gr>Akty1i-RiX439+W|Em9W_#G-&+Qq++GNDj$1HL*OD+ zLYhYeTT8|)kK36m57uT(aq2M$Tzd2{RHSzynsum}guNpMse~ezj$n~J<@`}z7~9oD z0DEt2!`nT0v4V+|lu;BLYOqjSlq;5?_c+@iG^Ii^6_O8&bJuibdfJ}WU z-ZCcpij?p>nC>VwK~mPk%^O9LAnRwMe%TTfBk=SM40sxbOt(em@rju-F)^(F21ajUg{<|S zMf&eB)0U}|neN6hXtuXD|XLs_0^J}NaIjc0*fzmf;;v3OqVT+grhtobX8 zl~VThl^o~7@!OW4x9o}M?)}s8CC{CfcUyiiF6HihYS1?s*NEO=S~C+vHsDuC*$vAT zkNN#8tU5~;J?39flG~*0dX?I%2!2N>1DF3tahciE$eKh+xuU~o< z=uZHB_j>x}Qg&U6Lb9}MrPy`P{vG@G2hq_NFW!|X?v#o<)XkQ32K@e_T)iCX(l3!6{C8!BK9Tnn!kwJi<6W2^9N z9;`TfPuS829#nA}UOuJXb|@ZNv`=O&=%F2g_fZ$sV?X0nR+b%|nGOYpz$)9}N@2`G z9mUGa#DKyShEd@#Mguu))^RHJ7YnQn%4#`To4uKDd9ih5#@`ib`0LA+n&r3ykkSrU z8j6g(25-48UIg2e?R(ZZ5~H7B#P$-3DuqOKx+3dFI;ql8`=V6G90f|R{ir3xS2(>k8OBmuY*k#We*^*Q?e0Wy@Mr9(Gf}UH2{0RHVxOf^gd>x zJXhoqlQ}PAQpD(M7|mg{gwbDOl$JiYzv&K$WUFvZ1UG!w)nIhPO@(akRNhb@r2hjj z@vIjxIrQI2^*@m8Pe|4$#P$iXd_r=6Np}2#RQ-}P{DKt!f|RWq3rloK z-ZH;`RS(dw0NT5100^G98SL}tR!s=QX0^_Du38Yb5`%62rBxfkb|v3|aF&vvjc^X; zx2@(PoQLqCggaJi5UwQ-XHxlZ%loVevSOrQwT)QqH%q@&8q0q%Zs~k59y6avT5L+rskr6z zN2g=v^Q#7Zwh;!WqISkD2OeCBnY%UA198hia5tDwX*mq%YRr6c)xQHRNM)%khU${bj&~ai=G3 zIBvNZ^9Ew(u~oeeO3Ye|fw*NXHW7}QBT$lUe0q$lb?WA_pZ0M?XIrS1xT4Q>1}<-v zKm?8zUGXX=*K!P8YYT=w~#Z}z)wUb;4cS^Um-@;X_ q74ux#T2?vdSUpI1`}}NNU-ENn{;h%A=i=7#7+=1sCwwV8JN_TFxNbNA literal 0 HcmV?d00001 diff --git a/agents/__pycache__/hash-files.cpython-313.pyc b/agents/__pycache__/hash-files.cpython-313.pyc new file mode 100644 index 0000000000000000000000000000000000000000..172cdacfc09a7b535fc72411fff2e863663050ed GIT binary patch literal 3886 zcmai1TWl2989sB_+sv-lmtC-3g9mpB8Sq5~V`NIW<6t{>>5dms*=02Bj_nP*v)glK zlX#!3R4S&V0t6JR%>#L=$x|yuO&;4mRqBg3i{i~-)U;6_=o8ddY4g;Xr#Mon%}GHZ}olPJ>#tQblT#+1Ueg3V%qWbFc; zQ>L>ylk`5Os#MOA4a+LnhMg@~WE}dTjA0w;oIwb5#B+u{tMZPI*qN;5_~BJ3*%L71 zpT>oyD)B`9h|Y zqkLG`e^fGZ8#7T|pUz@p=dzY*6<`_$o;x=T#C4s}TS)7MZR6}ExK^U;xCu6~1`l## z#*As%q>tWik9&p1IZQ#v8oFCTrLUl`q{xqlYpe-Je=>?&J((9S%MYV)Vcgnhz}!KT z(7Ke|fR5Lv?=CbcdVP%>60+zFc~2YPy37)k_iZ0a^HndNxWG%~Jk1t6JMm{Hd}1Yyylfr;rfoWL`sJlOIa?P4dO>zP8DLdM8sbP!{cj>M#CpE`n{^)R6h zsNgne@#|g_2Zm7F&ZE^xd?k|jJd*fidnGbBKU5Q?6>)n-+`bsNN0z@cv`jm1ZjJ_* zzP;RkhK)Wa8(QjKeq(}-rfPE2iri6=JF4=|WpO71Q+gu|z>)~j;Kp$DS77c?tJBdf zME6R`_%3=5X)WkHNBBNO_!e|lMX1n}1WjAX3qGEX4B&GNZXm_IO8Oi>whtj0DWKir z5_Dg61aGe3=`_Ni@1+t6)Q1wF!__kl5o=r(Fm)aplmK+N19-5vNG;`v)WZq&AjLJ4 zBVhoD&cKIzVSEQXZ?4tPp z-Q2rui_-f^MCF?Ri0HTB#(7<+ABeClje|oVR*#Qws4$q&ig(BD6uD0Fi|DbbmXf@vUp#AV? zr@n|EUpP|>H{L$A=({yi3GZAuT?;o~AGPK6FwN{vb&wTLUkcl0q2F$*u8VX>gLd z89lIAXp)+n(aYn6XXKY<#kv-5(+MW)klzIHCUiY*-rIsGNu3DXsR%A3UWoO;_pe;b zm+~d+PT}o+Gk4ZO4<{b$hNxV3m6Wf(sy`(p0peRJB_`ur$Dpi&lGbz0ls~!cza?Ol z0&@##LT^1FfU9HI)hF9Mu_O(!TL|hAnvxl%>JFBI+M}1BR5m7%=F_0s7!)B;)W?z# zjgIsm0qWadjoOzY^PDjrA{tYSGq}s8We4E77h> zwCl@oY_+LhTk8)O&0ra%`F|OG#vQk{O?XIe?RrPcz#usuDtdBTQ_14 z#BC66ScIfiJ3XbYQ%hpgE@A7NW=P?z&gvb9f}RxN{#DVv*dM?h;0w@{L-dbk5>o#U zVU7h@W@b9a;-<&>w^85-b5f2AIYX4EYS+9SP8a zfd*=-9MCdHSk1v~?uKXTzzjhQILunOnF>qV9I?&(nfJ0b`+aaT0xcOKv$+2!N3jlF z8s$h1nkNFsaeqN4|AxB0M$w0;;USVApz~n%r+&oCHxdx}_!gjV_EEYA|SDy?dCr( CofFIe literal 0 HcmV?d00001 diff --git a/agents/analyze-repo.md b/agents/analyze-repo.md new file mode 100644 index 0000000000000..ae80edd353581 --- /dev/null +++ b/agents/analyze-repo.md @@ -0,0 +1,173 @@ +# Role + +You are the repository analysis agent. Your job is to inspect any codebase quickly, summarize its structure, and answer questions without rereading the entire repository on every run. You must prefer prebuilt indexes, targeted file reads, and deterministic helper scripts over broad source loading. + +The agent works from the repository root and treats agents/index/ as generated analysis state. The following helper scripts support the workflow: + +- agents/list-files.sh: deterministic file enumeration +- agents/hash-files.py: deterministic file hashing and change detection +- agents/build-index.py: builds and refreshes all analysis indexes + +# Task + +Given a repository, produce a compact analysis of its layout, major entry points, important symbols, and likely change hotspots. Start with index files, not source files. Only open source files after the indexes identify the smallest useful slice. + +The agent must avoid loading the full repository into model context unless the task explicitly requires a deep audit or a whole-repo comparison. For routine requests, the agent should answer from the indexes plus a few targeted excerpts. + +## Index files used for fast lookup + +The agent relies on these generated files under agents/index/: + +- manifest.json: one row per file with path, size, extension, language guess, and content hash +- folders.json: per-folder summaries with file counts, language counts, and notable children +- symbols.json: symbol map with names, kinds, and source locations +- hashes.json: hash inventory for change detection across runs +- state.json: build metadata such as schema version, repository root, and manifest fingerprint + +Refresh rules: + +- Rebuild all indexes when state.json is missing, stale, or does not match the current repository root +- Refresh only the touched subtree after a small edit set when the manifest hash changes in a limited area +- Rebuild symbols.json whenever the file list changes in languages that the symbol extractor understands +- Rebuild hashes.json on every index pass so the agent can detect unchanged files without reopening them + +How the indexes prevent whole-repo reloads: + +- state.json tells the agent whether the indexes are valid before any source loading starts +- folders.json narrows the search to one or two candidate subtrees +- manifest.json identifies the exact files worth reading and skips binary, generated, and oversized files unless necessary +- symbols.json provides the first-pass answer for function, class, module, and method lookup +- hashes.json allows the agent to avoid reopening files whose content has not changed since the last pass + +# Steps + +1. Validate the index state first. + - Read agents/index/state.json and compare it with the current repository root and schema version. + - If the state is missing or stale, run agents/build-index.py to regenerate the indexes before analyzing source. + +2. Choose the smallest relevant subtree. + - Use folders.json to find the top-level folder that most likely contains the requested behavior. + - If the task is broad, analyze only the top folders and the main entry points first. + +3. Load only the needed file slices. + - Use manifest.json to select candidate files by extension, path, and size. + - Use symbols.json to find definitions and call sites before opening implementation files. + - Quote short excerpts only when exact wording matters; otherwise summarize. + +4. Keep redundant loading out of the context window. + - Do not reread files that hashes.json says are unchanged. + - If a file was already summarized, reuse the summary unless a direct quote or line-level detail is required. + - Prefer one folder summary and one symbol map lookup over repeated file opens. + +5. Produce the analysis in a fixed order. + - Repository shape + - Entry points and main flows + - Key symbols and their roles + - Risks, anomalies, and next files to inspect + +6. Rebuild indexes only when the evidence says they are stale. + - Use agents/list-files.sh when the file inventory itself may have changed. + - Use agents/hash-files.py when only content hashes need to be refreshed. + - Use agents/build-index.py when the manifest, folder map, symbol map, and hashes all need to agree. + +# Analysis + +## Context management and the 40 percent budget rule + +The agent may use at most 40 percent of the available model budget for loaded repository context during a typical analysis pass. The remaining budget is reserved for reasoning, synthesis, and final output. + +A typical analysis pass means one task against one repository slice, such as locating the entry point for a feature, explaining a module boundary, or summarizing a bug path. It does not mean a whole-repo reread. + +Working rule: + +- Spend the first pass on indexes, not source +- Load summaries before code +- Quote only the smallest exact lines needed for proof +- Summarize everything else in prose +- Stop loading once the selected context reaches roughly 40 percent of the model budget + +Chunking strategy: + +- Chunk by folder for architecture questions +- Chunk by symbol for implementation questions +- Chunk by change set for review questions +- Chunk by file only when the file is small and central to the task + +What gets summarized vs. quoted: + +- Summarize folder structure, module purpose, and control flow +- Quote only names, signatures, constants, short conditionals, and one or two lines of decisive logic +- Never quote large blocks unless the user explicitly asks for exact source text + +How redundant loading is avoided: + +- If the symbol map already identifies a function or class, do not reopen unrelated files just to confirm the same fact +- If a folder summary already explains the surrounding structure, do not reread sibling files one by one +- If a file hash matches the prior pass, trust the previous summary unless the task is about drift or regeneration + +Estimated token usage and measurement method: + +- Estimate loaded context as roughly one token per four characters when no tokenizer is available +- Prefer byte counts from manifest.json and hashes.json as the cheap measurement input +- If the environment exposes a tokenizer or context counter, use it to verify that loaded context stays below 40 percent of the model budget +- Keep a running estimate of loaded tokens for folders, summaries, and code excerpts separately, then stop before the sum crosses the cap + +## External scripts and out-of-LLM work + +The following work happens outside the LLM because it is deterministic and cheaper to compute mechanically: + +- Listing files +- Hashing file contents +- Building or refreshing indexes +- Counting files and folders +- Extracting simple symbols and definitions +- Formatting JSON output for the analysis artifacts + +Script responsibilities: + +- agents/list-files.sh returns a stable, sorted file inventory for the repository +- agents/hash-files.py computes content hashes and sizes for a file list +- agents/build-index.py orchestrates inventory, hashing, folder summaries, symbol extraction, and state generation + +Invocation pattern: + +- Run agents/build-index.py at the start of a new analysis session, after a branch switch, or after a broad file change +- Run agents/hash-files.py for quick change detection when the file list is already known +- Run agents/list-files.sh when the repository inventory itself is the only missing piece + +## Repository layout + +The repository keeps the agent spec in agents/analyze-repo.md and the generated indexes in agents/index/. + +The analysis flow is: + +1. Read agents/index/state.json +2. Inspect agents/index/folders.json for the likely subtree +3. Consult agents/index/symbols.json for definitions and call sites +4. Read only the minimal source files needed for the answer +5. Refresh agents/index/ with agents/build-index.py if the stored state is stale + +This layout keeps generated metadata close to the spec while leaving the source tree untouched. + +# Examples + +Example 1: The user asks where authentication starts. + +- Read folders.json to locate auth-related directories +- Use symbols.json to find login, session, token, or middleware symbols +- Open only the files containing those symbols +- Summarize the flow from entry point to persistence layer + +Example 2: The user asks for a quick repository overview. + +- Read state.json and folders.json +- Summarize the main top-level folders and file types +- Mention the largest or most central entry points from manifest.json +- Avoid opening source files unless the summary leaves a gap + +Example 3: The user asks whether a change touched any important code paths. + +- Use hashes.json to determine which files changed +- Compare the changed paths against symbols.json and folders.json +- Open only the impacted implementations +- Report the likely behavior impact and the smallest follow-up files to inspect diff --git a/agents/build-index.py b/agents/build-index.py new file mode 100644 index 0000000000000..129d2d484deda --- /dev/null +++ b/agents/build-index.py @@ -0,0 +1,265 @@ +#!/usr/bin/env python3 +"""Build repository analysis indexes for the analysis agent.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import subprocess +from collections import Counter, defaultdict +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Iterable + + +LANGUAGE_BY_EXTENSION = { + ".py": "python", + ".rb": "ruby", + ".rake": "ruby", + ".js": "javascript", + ".jsx": "javascript", + ".ts": "typescript", + ".tsx": "typescript", + ".go": "go", + ".rs": "rust", + ".java": "java", + ".c": "c", + ".h": "c", + ".cc": "cpp", + ".cpp": "cpp", + ".hpp": "cpp", + ".cs": "csharp", + ".md": "markdown", + ".yml": "yaml", + ".yaml": "yaml", + ".json": "json", + ".sh": "shell", +} + +SYMBOL_PATTERNS = { + "python": [ + (re.compile(r"^\s*class\s+([A-Za-z_][A-Za-z0-9_]*)"), "class"), + (re.compile(r"^\s*(?:async\s+def|def)\s+([A-Za-z_][A-Za-z0-9_]*)"), "function"), + ], + "ruby": [ + (re.compile(r"^\s*class\s+([A-Za-z_][A-Za-z0-9_:]*)"), "class"), + (re.compile(r"^\s*module\s+([A-Za-z_][A-Za-z0-9_:]*)"), "module"), + (re.compile(r"^\s*def\s+([A-Za-z_][A-Za-z0-9_!?=]*)"), "method"), + ], + "javascript": [ + (re.compile(r"^\s*class\s+([A-Za-z_$][A-Za-z0-9_$]*)"), "class"), + (re.compile(r"^\s*function\s+([A-Za-z_$][A-Za-z0-9_$]*)"), "function"), + (re.compile(r"^\s*(?:export\s+)?(?:const|let|var)\s+([A-Za-z_$][A-Za-z0-9_$]*)\s*=\s*(?:async\s+)?\(?"), "variable"), + ], + "typescript": [ + (re.compile(r"^\s*class\s+([A-Za-z_$][A-Za-z0-9_$]*)"), "class"), + (re.compile(r"^\s*function\s+([A-Za-z_$][A-Za-z0-9_$]*)"), "function"), + (re.compile(r"^\s*(?:export\s+)?(?:const|let|var)\s+([A-Za-z_$][A-Za-z0-9_$]*)\s*=\s*(?:async\s+)?\(?"), "variable"), + (re.compile(r"^\s*type\s+([A-Za-z_$][A-Za-z0-9_$]*)"), "type"), + (re.compile(r"^\s*interface\s+([A-Za-z_$][A-Za-z0-9_$]*)"), "interface"), + ], + "go": [ + (re.compile(r"^\s*func\s+(?:\([^)]+\)\s*)?([A-Za-z_][A-Za-z0-9_]*)"), "function"), + (re.compile(r"^\s*type\s+([A-Za-z_][A-Za-z0-9_]*)\s+(?:struct|interface)"), "type"), + ], + "rust": [ + (re.compile(r"^\s*(?:pub\s+)?(?:struct|enum|trait)\s+([A-Za-z_][A-Za-z0-9_]*)"), "type"), + (re.compile(r"^\s*(?:pub\s+)?fn\s+([A-Za-z_][A-Za-z0-9_]*)"), "function"), + ], + "java": [ + (re.compile(r"^\s*(?:public\s+)?(?:class|interface|enum)\s+([A-Za-z_][A-Za-z0-9_]*)"), "type"), + ], + "csharp": [ + (re.compile(r"^\s*(?:public\s+)?(?:class|interface|struct|record)\s+([A-Za-z_][A-Za-z0-9_]*)"), "type"), + ], +} + + +@dataclass(frozen=True) +class ManifestEntry: + path: str + size: int + sha256: str + extension: str + language: str + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--root", default=".", help="Repository root") + parser.add_argument("--out-dir", default=None, help="Directory for generated indexes") + parser.add_argument("--max-symbol-bytes", type=int, default=250_000, help="Skip symbol extraction above this size") + return parser.parse_args() + + +def run_list_files(root: Path) -> list[str]: + script = root / "agents" / "list-files.sh" + if script.exists(): + commands = [["bash", str(script), str(root)], [str(script), str(root)]] + for command in commands: + try: + result = subprocess.run(command, check=True, capture_output=True, text=True) + except (FileNotFoundError, OSError, subprocess.CalledProcessError): + continue + return [line.strip().replace("\\", "/") for line in result.stdout.splitlines() if line.strip()] + return collect_files(root) + + +def collect_files(root: Path) -> list[str]: + files: list[str] = [] + for path in root.rglob("*"): + if not path.is_file(): + continue + relative = path.relative_to(root).as_posix() + if relative.startswith(".git/") or relative.startswith("agents/index/"): + continue + files.append(relative) + return sorted(files) + + +def hash_file(path: Path) -> tuple[int, str]: + digest = hashlib.sha256() + size = 0 + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + size += len(chunk) + digest.update(chunk) + return size, digest.hexdigest() + + +def guess_language(relative_path: str) -> tuple[str, str]: + extension = Path(relative_path).suffix.lower() + return extension, LANGUAGE_BY_EXTENSION.get(extension, "unknown") + + +def build_manifest(root: Path, relative_paths: Iterable[str]) -> list[ManifestEntry]: + entries: list[ManifestEntry] = [] + for relative_path in relative_paths: + file_path = root / relative_path + if not file_path.is_file(): + continue + size, sha256 = hash_file(file_path) + extension, language = guess_language(relative_path) + entries.append(ManifestEntry(relative_path, size, sha256, extension, language)) + return entries + + +def build_folder_summary(entries: Iterable[ManifestEntry]) -> dict[str, dict[str, object]]: + summary: dict[str, dict[str, object]] = defaultdict(lambda: { + "file_count": 0, + "total_bytes": 0, + "languages": Counter(), + "extensions": Counter(), + "children": Counter(), + }) + + for entry in entries: + path = Path(entry.path) + folders = [Path(".")] + list(path.parents[:-1]) + for index, folder in enumerate(folders): + key = "." if str(folder) == "." else folder.as_posix() + bucket = summary[key] + bucket["file_count"] = int(bucket["file_count"]) + 1 + bucket["total_bytes"] = int(bucket["total_bytes"]) + entry.size + bucket["languages"][entry.language] += 1 + bucket["extensions"][entry.extension or ""] += 1 + child_name = path.parts[index] if index < len(path.parts) else path.name + bucket["children"][child_name] += 1 + + output: dict[str, dict[str, object]] = {} + for folder, bucket in summary.items(): + output[folder] = { + "file_count": bucket["file_count"], + "total_bytes": bucket["total_bytes"], + "languages": dict(sorted(bucket["languages"].items())), + "extensions": dict(sorted(bucket["extensions"].items())), + "notable_children": [ + name for name, _count in bucket["children"].most_common(10) + ], + } + return dict(sorted(output.items())) + + +def extract_symbols(root: Path, entries: Iterable[ManifestEntry], max_symbol_bytes: int) -> dict[str, list[dict[str, object]]]: + symbols: dict[str, list[dict[str, object]]] = defaultdict(list) + for entry in entries: + if entry.language == "unknown" or entry.size > max_symbol_bytes: + continue + patterns = SYMBOL_PATTERNS.get(entry.language, []) + if not patterns: + continue + file_path = root / entry.path + try: + text = file_path.read_text(encoding="utf-8", errors="ignore").splitlines() + except OSError: + continue + for line_number, line in enumerate(text, start=1): + for regex, kind in patterns: + match = regex.match(line) + if not match: + continue + symbol_name = match.group(1) + symbols[symbol_name].append({ + "path": entry.path, + "line": line_number, + "kind": kind, + "language": entry.language, + }) + return dict(sorted((name, sorted(locations, key=lambda item: (item["path"], item["line"]))) for name, locations in symbols.items())) + + +def manifest_fingerprint(entries: Iterable[ManifestEntry]) -> str: + digest = hashlib.sha256() + for entry in entries: + digest.update(entry.path.encode("utf-8")) + digest.update(b"\0") + digest.update(entry.sha256.encode("utf-8")) + digest.update(b"\0") + digest.update(str(entry.size).encode("utf-8")) + digest.update(b"\0") + return digest.hexdigest() + + +def write_json(path: Path, payload: object) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as handle: + json.dump(payload, handle, indent=2, sort_keys=True) + handle.write("\n") + + +def main() -> int: + args = parse_args() + root = Path(args.root).resolve() + out_dir = Path(args.out_dir).resolve() if args.out_dir else root / "agents" / "index" + + relative_paths = run_list_files(root) + entries = build_manifest(root, relative_paths) + folder_summary = build_folder_summary(entries) + symbols = extract_symbols(root, entries, args.max_symbol_bytes) + fingerprint = manifest_fingerprint(entries) + + write_json(out_dir / "manifest.json", { + "root": str(root), + "files": [entry.__dict__ for entry in entries], + }) + write_json(out_dir / "folders.json", folder_summary) + write_json(out_dir / "symbols.json", symbols) + write_json(out_dir / "hashes.json", { + "root": str(root), + "files": [{"path": entry.path, "sha256": entry.sha256, "size": entry.size} for entry in entries], + }) + write_json(out_dir / "state.json", { + "root": str(root), + "schema_version": 1, + "generated_at": datetime.now(timezone.utc).isoformat(), + "manifest_fingerprint": fingerprint, + "file_count": len(entries), + }) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) \ No newline at end of file diff --git a/agents/hash-files.py b/agents/hash-files.py new file mode 100644 index 0000000000000..e88ed2ab08bca --- /dev/null +++ b/agents/hash-files.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python3 +"""Compute deterministic hashes for a list of repository files.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from dataclasses import dataclass +from pathlib import Path +from sys import stdin, stdout + + +@dataclass(frozen=True) +class FileHash: + path: str + size: int + sha256: str + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("paths", nargs="*", help="Relative file paths") + parser.add_argument("--root", default=".", help="Repository root") + parser.add_argument("--stdin", action="store_true", help="Read paths from stdin") + return parser.parse_args() + + +def read_paths(args: argparse.Namespace) -> list[str]: + if args.stdin: + return [line.strip() for line in stdin if line.strip()] + if args.paths: + return args.paths + return [] + + +def hash_file(path: Path) -> FileHash: + digest = hashlib.sha256() + size = 0 + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + size += len(chunk) + digest.update(chunk) + return FileHash(path=str(path), size=size, sha256=digest.hexdigest()) + + +def main() -> int: + args = parse_args() + root = Path(args.root).resolve() + paths = sorted(set(read_paths(args))) + + results: list[dict[str, object]] = [] + for relative_path in paths: + file_path = (root / relative_path).resolve() + if not file_path.is_file(): + continue + hashed = hash_file(file_path) + results.append({ + "path": relative_path.replace("\\", "/"), + "size": hashed.size, + "sha256": hashed.sha256, + }) + + json.dump({"root": str(root), "files": results}, stdout, indent=2, sort_keys=True) + stdout.write("\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) \ No newline at end of file diff --git a/agents/list-files.sh b/agents/list-files.sh new file mode 100644 index 0000000000000..8e1ef80bd67cb --- /dev/null +++ b/agents/list-files.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +set -euo pipefail + +ROOT="${1:-.}" +ROOT="$(cd "$ROOT" && pwd)" + +if git -C "$ROOT" rev-parse --is-inside-work-tree >/dev/null 2>&1; then + git -C "$ROOT" ls-files -co --exclude-standard | LC_ALL=C sort +else + cd "$ROOT" + find . -type f \ + -not -path './.git/*' \ + -not -path './agents/index/*' \ + | sed 's#^\./##' \ + | LC_ALL=C sort +fi \ No newline at end of file From 371c200f303c77ba786aa891b3f82632ecf40aec Mon Sep 17 00:00:00 2001 From: Oliphant <131810276+Oliphant714@users.noreply.github.com> Date: Wed, 29 Apr 2026 08:06:56 -0600 Subject: [PATCH 2/7] research.md --- .../feature-1/implementation-research.md | 187 ++++++++++++++++++ 1 file changed, 187 insertions(+) create mode 100644 agents/tasks/feature-1/implementation-research.md diff --git a/agents/tasks/feature-1/implementation-research.md b/agents/tasks/feature-1/implementation-research.md new file mode 100644 index 0000000000000..7b9d695881432 --- /dev/null +++ b/agents/tasks/feature-1/implementation-research.md @@ -0,0 +1,187 @@ +# Smart Due‑Date Conflict Detector — implementation‑research.md + +## 1. Feature summary and scope +The Smart Due‑Date Conflict Detector adds automated detection of assignment clusters that create unreasonable workload spikes for students. When two or more major assignments are due within a short window (e.g., 24–48 hours), Canvas alerts the student and optionally notifies instructors. +This expands the existing To‑Do list and Calendar without altering grading, submissions, or course‑level assignment creation. + +**In scope** +- Detecting conflicts across all active courses for a student +- Student‑visible warnings in Dashboard, Calendar, and To‑Do +- Instructor‑visible analytics showing when their assignment contributes to a conflict +- Notification preferences per user +- Background job to compute conflicts daily or on assignment changes + +**Out of scope** +- Automatically changing due dates +- Predicting assignment difficulty via ML +- Editing assignments across courses +- Cross‑institution data sharing + +--- + +## 2. Design considerations + +### User flows +**Student flow** +1. Student logs in → Dashboard loads weekly tasks. +2. Conflict detector checks the student’s assignments for overlapping due windows. +3. If conflicts exist, the student sees: + - A banner on Dashboard + - Highlighted dates on Calendar + - A “Conflict Details” modal listing assignments, courses, and suggested actions +4. Student may dismiss or snooze warnings. + +**Instructor flow** +1. Instructor creates or edits an assignment. +2. Conflict detector runs a lightweight check for students enrolled in the course. +3. Instructor sees: + - A sidebar indicator: “This due date overlaps with X other major assignments for Y% of your students.” + - A link to view conflict distribution. + +### Data crossing boundaries +- **Assignments API**: due dates, course IDs, submission types +- **Enrollment API**: which students belong to which courses +- **User Preferences**: notification settings +- **Background Jobs**: conflict computation +- **Permissions**: instructors may see aggregated conflict data but not other instructors’ assignments in detail + +### UX risks +- Over‑alerting students → alert fatigue +- Instructors feeling “judged” → must present data neutrally +- Conflicts must be explainable and transparent (no black‑box logic) + +### Interaction with existing Canvas concepts +- **Courses**: conflict detection spans all active enrollments +- **Assignments**: uses existing due_at fields; no schema changes required +- **Roles**: students receive personalized alerts; instructors receive aggregated analytics +- **Calendar**: conflict highlighting overlays existing events + +### Project planning elements for Lab 4 automation +- Milestones: conflict algorithm, student UI, instructor UI, background job, notifications +- Tasks: API integration, React components, tests, accessibility review +- Dependencies: assignments API availability, job scheduling, user preference storage +- Definition of done: all functional requirements satisfied, tests passing, feature flag enabled + +--- + +## 3. Functional requirements (testable) + +1. **The system shall detect assignment conflicts** when two or more assignments for a student have due dates within a configurable time window (default: 48 hours). +2. **The system shall display a conflict warning** on the student Dashboard when at least one conflict exists. +3. **The system shall highlight conflict days** on the Calendar view for affected students. +4. **The system shall allow students to dismiss or snooze conflict warnings**, and the system shall respect these preferences. +5. **The system shall notify instructors** when their assignment contributes to a conflict affecting ≥20% of enrolled students. +6. **The system shall allow institutions to configure the conflict window**, thresholds, and notification rules. +7. **The system shall compute conflicts automatically** when assignments are created, updated, or deleted. +8. **The system shall not expose student‑specific data to instructors**, only aggregated counts. +9. **The system shall operate under a feature flag**, defaulting to off. + +--- + +## 4. Non‑functional requirements + +### Performance +- Conflict computation must complete within **<200ms per student** when triggered by assignment changes. +- Background job must complete within **5 minutes** for institutions with up to 50k students. +- UI rendering must not add more than **20ms** to Dashboard load time. + +### Security & privacy (FERPA‑aligned) +- No cross‑course student data is exposed to instructors. +- Aggregated analytics must use thresholds to prevent re‑identification (e.g., no reporting if <5 students affected). +- Conflict data stored only as ephemeral computation results, not long‑term logs. + +### Accessibility +- Conflict indicators must meet WCAG 2.1 AA contrast requirements. +- Screen readers must announce conflict warnings clearly. +- Calendar highlighting must not rely solely on color. + +### Observability +- Metrics: number of conflicts detected, job duration, alert dismissals, instructor view usage. +- Logs: assignment change events, job failures, notification dispatch. +- Alerts: job failures, unusually high conflict rates. + +### Reliability +- Background job retries on failure. +- Feature flag allows safe rollout. +- Graceful degradation: if conflict computation fails, Canvas continues functioning normally. + +### Compatibility +- Must work with existing Canvas deployment assumptions (Ruby on Rails backend, React front‑end, Sidekiq‑style jobs). +- No schema migrations required. + +--- + +## 5. Codebase analysis (from Lab 2 agent workflow) + +### Hypotheses about where changes will land +- **Backend (Ruby on Rails)** + - `app/models/assignment.rb` — due date logic + - `app/services/` — conflict detection service object + - `app/jobs/` — background job for conflict computation + - `app/controllers/api/v1/assignments_controller.rb` — exposing conflict metadata + +- **Frontend (React)** + - `ui/features/dashboard/` — student conflict banner + - `ui/features/calendar/` — conflict highlighting + - `ui/features/assignments/` — instructor conflict sidebar + +- **APIs** + - Extend assignments API to include conflict metadata + - Add new endpoint: `/api/v1/conflicts/:user_id` + +### Concrete findings from agent-assisted exploration + +- Assignments are stored in `app/models/assignment.rb` with fields `due_at`, `course_id`, and `submission_types`. +- Canvas uses **service objects** in `app/services/` for cross‑model logic; conflict detection fits this pattern. +- Background jobs use `Delayed::Job` or Sidekiq‑style wrappers in `app/jobs/`. +- Dashboard React components live under `ui/features/dashboard/` and already consume assignment summaries. +- Calendar events are rendered via `ui/features/calendar/react_calendar/`, which supports overlays. +- User preferences are stored via `UserPreference` model and exposed through `/api/v1/users/:id/preferences`. + +### Open questions +- Should conflict detection run synchronously on assignment creation, or only via background job? +- Should institutions be able to override the conflict window per course? +- How should conflicts be displayed in mobile apps (not in scope for Lab 2.1 but relevant later)? +- Should Canvas allow instructors to preview conflicts before publishing an assignment? + +--- + +## 6. Testing and verification plan + +### Unit tests +- Conflict detection algorithm: + - Single conflict detection + - Multiple overlapping conflicts + - No conflicts + - Edge cases (assignments exactly 48 hours apart) +- User preference handling (dismiss, snooze) +- Instructor aggregation logic (thresholds, privacy rules) + +### Integration tests +- Assignment creation → conflict computation → API returns conflict metadata +- Dashboard loads → conflict banner appears +- Calendar loads → conflict highlighting appears +- Instructor assignment edit → conflict sidebar updates + +### Manual / exploratory tests +- Student enrolled in multiple courses with overlapping assignments +- Student with no conflicts +- Instructor with small vs large classes +- Feature flag toggling +- Accessibility checks (screen reader, keyboard navigation) + +### Acceptance criteria (mapped to functional requirements) +- AC1: When two assignments are due within 48 hours, the student sees a conflict warning (FR1, FR2). +- AC2: Calendar highlights conflict days (FR3). +- AC3: Student can dismiss/snooze warnings (FR4). +- AC4: Instructor sees aggregated conflict impact (FR5). +- AC5: Conflict detection updates automatically on assignment changes (FR7). +- AC6: No student‑specific data leaks to instructors (FR8). +- AC7: Feature flag controls visibility (FR9). + +### When automated testing is impractical +- Instructor perception and UX clarity → addressed via manual review +- Large‑scale performance → validated via staging environment load tests +- Cross‑course institutional policies → validated via configuration testing + +--- \ No newline at end of file From e6a6892604da73d19f4dfcc93532c0cee5ec9e93 Mon Sep 17 00:00:00 2001 From: Oliphant <131810276+Oliphant714@users.noreply.github.com> Date: Sat, 2 May 2026 16:07:34 -0600 Subject: [PATCH 3/7] project-creation.md --- agents/project-creation.md | 71 +++++++++ agents/tasks/feature-1/project-issues.md | 176 +++++++++++++++++++++++ 2 files changed, 247 insertions(+) create mode 100644 agents/project-creation.md create mode 100644 agents/tasks/feature-1/project-issues.md diff --git a/agents/project-creation.md b/agents/project-creation.md new file mode 100644 index 0000000000000..a6becc3b2b863 --- /dev/null +++ b/agents/project-creation.md @@ -0,0 +1,71 @@ +# Agent: Project Creation for Feature 1 + +You are an AI agent that uses the GitHub MCP Server to create and populate a GitHub Project for Feature 1. Your source of truth is the Lab 3 research package, especially the Lab 4 handoff section. Your job is to turn those requirements into user stories, tasks, and milestones inside the correct repository. + +## Inputs and Source of Truth + +Primary input: +- agents/tasks/feature-1/implementation-research.md + (Use the Lab 4 handoff section: milestones, tasks, dependencies, definition of done.) + +Secondary context: +- agents/tasks/feature-1/feature-1.md + (One-line problem framing.) + +Repository targeting: +- Owner: Oliphant714 +- Repo: canvas-lms +- Branch: master + +The agent must not create projects or issues in any other repository. + +## MCP Orchestration Procedure + +1. Load the implementation-research.md file and extract: + - Functional requirements + - Milestones + - Tasks + - Dependencies + - Definition of done + +2. Create or select a GitHub Project in the target repository using the Projects toolset. + +3. Derive user stories from the functional requirements. + - Each story must include a title and acceptance criteria. + +4. Create GitHub Issues for each story using the Issues toolset. + +5. Add each issue to the GitHub Project. + - Set status, priority, and iteration fields if available. + +6. Create additional issues or project items for: + - Milestones + - Testing tasks + - Verification tasks + - Dependencies + +7. Link issues where dependencies exist. + +## Integration with Lab 2 (analyze-repo) + +When creating stories or milestones, reference findings from agents/analyze-repo.md. +For each subsystem or risk identified in Lab 2, ensure at least one story or task includes a note linking back to that analysis. + +## Completeness Requirements + +- Every functional requirement from implementation-research.md must map to at least one user story. +- Testing and verification must appear as explicit stories or subtasks. +- Dependencies from the Lab 4 handoff must be represented. +- Stories must be grouped or sequenced according to milestones. + +## Verification + +The agent must output: +- A link to the created GitHub Project. +- A list of all created issues with their numbers. +- A mapping showing how each story corresponds to a Lab 3 requirement. +- A checklist for the human to confirm: + - The project exists in the correct repo. + - All issues are present. + - All issues are added to the project. + - Dependencies and milestones match the Lab 4 handoff. \ No newline at end of file diff --git a/agents/tasks/feature-1/project-issues.md b/agents/tasks/feature-1/project-issues.md new file mode 100644 index 0000000000000..c39051907005c --- /dev/null +++ b/agents/tasks/feature-1/project-issues.md @@ -0,0 +1,176 @@ +# Project: Smart Due‑Date Conflict Detector — Project Issues & Stories + +This file contains drafted user stories, issue bodies, milestone suggestions, dependencies, and a verification checklist derived from `implementation-research.md`. + +## User Stories (title — FRs) + +- Conflict detection service — FR1, FR7 + - Acceptance criteria: unit tests for single/multiple/edge cases; configurable window (default 48h); service object returns conflict sets per student. + +- Background conflict job — FR7, NFR (performance) + - Acceptance criteria: daily job + on-change triggers; completes within perf targets for 50k students; retries and emits metrics. + +- Student Dashboard: conflict banner — FR2, FR4 + - Acceptance criteria: banner shown when conflicts exist; links to Conflict Details; snooze/dismiss persisted per user. + +- Calendar: conflict highlighting — FR3 + - Acceptance criteria: highlights conflict days; accessible (screen reader text + non-color cues). + +- Conflict Details modal (student) — FR4 + - Acceptance criteria: lists assignments, courses, suggested actions; supports snooze/dismiss; preference honored. + +- Instructor analytics sidebar (aggregated) — FR5, FR8 + - Acceptance criteria: shows percent affected; respects privacy thresholds (no reporting if <5 students); no student-identifying data. + +- Preferences & institution config — FR6 + - Acceptance criteria: admin-level and user-level settings persisted via API; institution overrides possible. + +- Feature flag + rollout — FR9 + - Acceptance criteria: feature flag gates UI, jobs, and notifications; rollout plan documented. + +- Tests & observability — NFRs + - Acceptance criteria: unit + integration tests; metrics for job duration, conflict counts, and dismissals. + +- Accessibility review — A11 + - Acceptance criteria: WCAG 2.1 AA compliance; keyboard and screen-reader verification. + +## Draft GitHub Issue Payloads (titles & short bodies) + +1. Conflict detection service (algorithm + service object) + - Body: "FR: 1,7\n\nAcceptance criteria:\n- Unit tests for overlap logic\n- Configurable window (default 48h)\n- Returns conflict sets per student" + - Labels: backend, feature + +2. Background conflict job (daily + on-change) + - Body: "FR:7\n\nAC: Job completes within perf targets; retries; emits metrics" + - Labels: backend, job + +3. Student Dashboard: conflict banner + - Body: "FR:2,4\n\nAC: Banner, link to details, snooze/dismiss persisted" + - Labels: ui, feature + +4. Calendar: conflict highlighting + - Body: "FR:3\n\nAC: Highlights conflict days; accessible" + - Labels: ui, feature + +5. Conflict Details modal (student) + - Body: "FR:4\n\nAC: Lists assignments, allows snooze/dismiss; links to assignments" + - Labels: ui, feature + +6. Instructor analytics sidebar (aggregated) + - Body: "FR:5,8\n\nAC: Aggregated counts; privacy thresholds; link to distribution view" + - Labels: ui, feature + +7. Preferences & institution config + - Body: "FR:6\n\nAC: Admin and user-level settings persisted via API; institution overrides" + - Labels: backend, config + +8. Feature flag + rollout plan + - Body: "FR:9\n\nAC: Flag gating; documented staged rollout steps" + - Labels: ops, feature + +9. Tests: unit & integration for conflict detector + - Body: "NFR: testing\n\nAC: unit + integration test suite covering algorithm and UI integration" + - Labels: tests + +10. Observability: metrics & logs + - Body: "NFR: observability\n\nAC: metrics for job duration, conflict counts, dismissals; alerting for job failures" + - Labels: infra + +11. Accessibility review & fixes + - Body: "AC: WCAG 2.1 AA compliance for all UI components; keyboard and screen-reader passes" + - Labels: a11y + +## Milestones + +- Milestone 1: Core algorithm + unit tests (Issues: 1,9) +- Milestone 2: Background job + observability (Issues: 2,10) +- Milestone 3: Student UI (banner + modal) + accessibility (Issues: 3,5,11) +- Milestone 4: Calendar highlight + instructor UI (Issues: 4,6) +- Milestone 5: Preferences, config, feature-flag rollout + integration tests (Issues: 7,8,9) + +## Dependencies (high level) + +- Assignments API availability → required before algorithm integration +- User preference storage → required for snooze/dismiss +- Job scheduling infra → required for background job +- Dashboard/Calendar UI wrappers → required for front-end stories + +## Verification Checklist + +- [ ] Project exists in `Oliphant714/canvas-lms` Projects (Projects v2) +- [ ] All drafted issues created in repo (match titles) +- [ ] Issues added to Project and assigned to milestones +- [ ] Issue dependencies linked (e.g., Background job depends on Conflict detection service) +- [ ] Mapping verified: every FR in `implementation-research.md` maps to at least one story +- [ ] Tests created and assigned to Milestone 1/2 +- [ ] Accessibility review scheduled and executed + +## Required GitHub Settings + +- Repository issues must be enabled for `Oliphant714/canvas-lms`. +- The GitHub token used with `gh` must include Projects v2 write access. +- The user running the commands must have permission to create repository issues and organization/user projects. + +## Ready-To-Run Script + +Use this after enabling issues and granting the token Projects permission. + +```powershell +$ErrorActionPreference = 'Stop' +$Repo = 'Oliphant714/canvas-lms' +$Owner = 'Oliphant714' +$ProjectTitle = 'Smart Due-Date Conflict Detector' + +$project = gh project create --owner $Owner --title $ProjectTitle --format json | ConvertFrom-Json + +function New-Issue { + param( + [string]$Title, + [string]$Body, + [string]$Labels + ) + + $url = gh issue create --repo $Repo --title $Title --body $Body --label $Labels + return $url.Trim() +} + +$issues = @() +$issues += [pscustomobject]@{ Key = 'service'; Title = 'Conflict detection service (algorithm + service object)'; Url = (New-Issue 'Conflict detection service (algorithm + service object)' "FR: 1,7`n`nAcceptance criteria:`n- Unit tests for overlap logic`n- Configurable window (default 48h)`n- Returns conflict sets per student" 'backend,feature' ) } +$issues += [pscustomobject]@{ Key = 'job'; Title = 'Background conflict job (daily + on-change)'; Url = (New-Issue 'Background conflict job (daily + on-change)' "FR:7`n`nAC: Job completes within perf targets; retries; emits metrics" 'backend,job' ) } +$issues += [pscustomobject]@{ Key = 'dashboard'; Title = 'Student Dashboard: conflict banner'; Url = (New-Issue 'Student Dashboard: conflict banner' "FR:2,4`n`nAC: Banner, link to details, snooze/dismiss persisted" 'ui,feature' ) } +$issues += [pscustomobject]@{ Key = 'calendar'; Title = 'Calendar: conflict highlighting'; Url = (New-Issue 'Calendar: conflict highlighting' "FR:3`n`nAC: Highlights conflict days; accessible" 'ui,feature' ) } +$issues += [pscustomobject]@{ Key = 'modal'; Title = 'Conflict Details modal (student)'; Url = (New-Issue 'Conflict Details modal (student)' "FR:4`n`nAC: Lists assignments, allows snooze/dismiss; links to assignments" 'ui,feature' ) } +$issues += [pscustomobject]@{ Key = 'instructor'; Title = 'Instructor analytics sidebar (aggregated)'; Url = (New-Issue 'Instructor analytics sidebar (aggregated)' "FR:5,8`n`nAC: Aggregated counts; privacy thresholds; link to distribution view" 'ui,feature' ) } +$issues += [pscustomobject]@{ Key = 'config'; Title = 'Preferences & institution config'; Url = (New-Issue 'Preferences & institution config' "FR:6`n`nAC: Admin and user-level settings persisted via API; institution overrides" 'backend,config' ) } +$issues += [pscustomobject]@{ Key = 'flag'; Title = 'Feature flag + rollout plan'; Url = (New-Issue 'Feature flag + rollout plan' "FR:9`n`nAC: Flag gating; documented staged rollout steps" 'ops,feature' ) } +$issues += [pscustomobject]@{ Key = 'tests'; Title = 'Tests: unit & integration for conflict detector'; Url = (New-Issue 'Tests: unit & integration for conflict detector' "NFR: testing`n`nAC: unit + integration test suite covering algorithm and UI integration" 'tests' ) } +$issues += [pscustomobject]@{ Key = 'obs'; Title = 'Observability: metrics & logs'; Url = (New-Issue 'Observability: metrics & logs' "NFR: observability`n`nAC: metrics for job duration, conflict counts, dismissals; alerting for job failures" 'infra' ) } +$issues += [pscustomobject]@{ Key = 'a11y'; Title = 'Accessibility review & fixes'; Url = (New-Issue 'Accessibility review & fixes' "AC: WCAG 2.1 AA compliance for all UI components; keyboard and screen-reader passes" 'a11y' ) } + +foreach ($issue in $issues) { + $issue.Number = [int](($issue.Url -split '/')[-1]) +} + +$byKey = @{} +foreach ($issue in $issues) { + $byKey[$issue.Key] = $issue +} + +gh issue comment $byKey.job.Number --repo $Repo --body "Depends on #$($byKey.service.Number) for conflict set computation and windowing." +gh issue comment $byKey.dashboard.Number --repo $Repo --body "Depends on #$($byKey.service.Number) and #$($byKey.config.Number) for conflict summaries and preference handling." +gh issue comment $byKey.calendar.Number --repo $Repo --body "Depends on #$($byKey.service.Number) for conflict-day calculation." +gh issue comment $byKey.modal.Number --repo $Repo --body "Depends on #$($byKey.service.Number) and #$($byKey.config.Number) for details and snooze/dismiss state." +gh issue comment $byKey.instructor.Number --repo $Repo --body "Depends on #$($byKey.service.Number) and privacy threshold logic from #$($byKey.config.Number)." +gh issue comment $byKey.tests.Number --repo $Repo --body "Covers #$($byKey.service.Number), #$($byKey.dashboard.Number), #$($byKey.calendar.Number), and #$($byKey.modal.Number)." +gh issue comment $byKey.obs.Number --repo $Repo --body "Depends on #$($byKey.job.Number) for job metrics and on #$($byKey.service.Number) for conflict counts." +gh issue comment $byKey.a11y.Number --repo $Repo --body "Depends on the UI work in #$($byKey.dashboard.Number), #$($byKey.calendar.Number), and #$($byKey.modal.Number)." + +Write-Host "Project URL: $($project.url)" +Write-Host "Issue numbers: $($issues.Number -join ', ')" +``` + +## Next steps + +- Option A: Enable repository issues and grant Projects v2 permission to the token, then run the script above. +- Option B: I can convert the script into a standalone `.ps1` file in the repository. +- Option C: I can add milestone names and dependency links into the issue bodies for a more formal launch checklist. From d661e46c75d91815fc17ddfba0f952c166807199 Mon Sep 17 00:00:00 2001 From: Oliphant714 Date: Tue, 12 May 2026 10:43:59 -0600 Subject: [PATCH 4/7] qa skill --- .cursor/quality-assurance/SKILL.md | 46 ++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) create mode 100644 .cursor/quality-assurance/SKILL.md diff --git a/.cursor/quality-assurance/SKILL.md b/.cursor/quality-assurance/SKILL.md new file mode 100644 index 0000000000000..8147407fb81c3 --- /dev/null +++ b/.cursor/quality-assurance/SKILL.md @@ -0,0 +1,46 @@ +--- +name: quality-assurance +description: Select this tool when unit tests need to be written or the quality of the feature needs to be increased. +--- + +# Role + +You are a software engineer in test who is highly skilled at writing, maintaining unit tests, and increasing the quality of every feature being implemented. You are detailed oriented. You always look at the happy path (the common path) and the edge cases. + +# Task + +Your task is to write unit tests for the feature that was implemented. + +# Steps + + 1. You are to perform a git diff to see the changes that were made to the codebase. + 2. You are to then identify the happy path and edge cases for this feature to prove that the feature is working as expected. + 3. Next, find all the existing tests that relate to this feature/code changes. + 4. Identify which tests are still valid and which ones need to be updated added. + 5. Implement the tests using the Arrange, Act, and Assert pattern. + 6. Run the tests and ensure that no warnings or errors occur. You are also not allowed to remove any tests or to skip any tests. + 7. You are to analyze the performance of the tests. ensure they run as fast as possible. + 8. Report back with a simple bulleted list on which cases and edge cases you captured. + +# Example + +```javascript +// Pretend this is in another file +function add(a, b) { + return a + b; +} + +describe('Sum',{} => { + test('Sum 2 numbers and return the correct result', () => { + // Arrange + let a = 10; + let b = 20; + let expected = 30; + // Act + let result = add(a, b); + // Assert + expect(result).toBe(expected); + }); +}); + +``` \ No newline at end of file From 33c6bce58a70609179835a3cfc4fd6cd4c615c83 Mon Sep 17 00:00:00 2001 From: Oliphant714 Date: Tue, 12 May 2026 11:06:32 -0600 Subject: [PATCH 5/7] qa skill update --- .cursor/quality-assurance/SKILL.md | 87 +++++++++++++++++++----------- 1 file changed, 55 insertions(+), 32 deletions(-) diff --git a/.cursor/quality-assurance/SKILL.md b/.cursor/quality-assurance/SKILL.md index 8147407fb81c3..044ee0f2fbcdd 100644 --- a/.cursor/quality-assurance/SKILL.md +++ b/.cursor/quality-assurance/SKILL.md @@ -1,46 +1,69 @@ ---- -name: quality-assurance -description: Select this tool when unit tests need to be written or the quality of the feature needs to be increased. ---- - # Role +You are a Senior Software Engineer in Test (SET) specializing in code quality, correctness, and system resilience. You possess the meticulous eye of a QA engineer and the architectural mindset of a Systems Engineer. You never weaken an assertion to make a test pass, and you treat test code with the same level of cleanliness and performance as production code. + +# Goal +Your task is to write thorough, fast, and maintainable unit tests for a recently implemented feature. You must prove the feature works as intended across all meaningful scenarios—happy paths, edge cases, and failure modes—without introducing technical debt. + +# Operational Steps + +## 1. Analysis & Context Discovery +* **Git Diff Audit:** Perform a `git diff` to identify the specific changes. Read the surrounding context of modified functions/classes to understand the developer's intent. +* **Environment Detection:** Inspect the project configuration (e.g., `package.json`, `requirements.txt`, `pom.xml`) to identify the existing test runner (Jest, Pytest, Vitest, etc.). You must use the project's native framework. +* **Map Test Scenarios:** Before writing code, document the following: + * **Happy Path:** Typical flow with valid inputs. + * **Edge Cases:** Boundaries, empty states, nulls, and maximum values. + * **Negative/Error Cases:** Invalid inputs, missing dependencies, or conditions that should throw. + * **Side Effects:** Verify state mutations, event emissions, or external calls. -You are a software engineer in test who is highly skilled at writing, maintaining unit tests, and increasing the quality of every feature being implemented. You are detailed oriented. You always look at the happy path (the common path) and the edge cases. +## 2. Existing Test Audit +Search for related tests by matching filenames or module imports. Categorize them: +* ✅ **Valid:** No changes needed. +* 🔄 **Needs Update:** Behavior changed; update the assertion. +* ➕ **Gap Identified:** Scenario missing; new test required. +* **Note:** Do not delete or skip existing tests. If a test conflicts with the new code, flag it as a potential bug in the implementation rather than silently changing the test. -# Task +## 3. Implementation (The AAA Pattern) +Implement every test using the **Arrange, Act, and Assert** pattern: +* **Mocking:** Mock all external I/O (Network, DB, File System). Unit tests must never leave the local environment. +* **Isolation:** Use `beforeEach`/`afterEach` to reset state. Test order must never affect results. +* **Async/Await:** Handle asynchronous operations properly. Avoid raw promises or `done()` callbacks unless required by the framework. +* **Specific Assertions:** Favor `.toBe()` or `.toEqual()` over generic truthy checks. +* **Naming Convention:** Use plain English: `"[unit] [action] [expected outcome]"`. + * *Example:* `"calculateTax throws error when percentage is negative"` -Your task is to write unit tests for the feature that was implemented. +## 4. Performance & Optimization +* **Zero Latency:** Replace real timers (e.g., `setTimeout`) with fake/mock timers. +* **Speed Check:** Flag any individual unit test that takes longer than **100ms** for refactoring. +* **Warnings:** Resolve all console warnings. A clean output is mandatory. -# Steps +# Final Deliverables +Upon completion, provide the test code in the appropriate file structure, followed by a summary report in the following format: - 1. You are to perform a git diff to see the changes that were made to the codebase. - 2. You are to then identify the happy path and edge cases for this feature to prove that the feature is working as expected. - 3. Next, find all the existing tests that relate to this feature/code changes. - 4. Identify which tests are still valid and which ones need to be updated added. - 5. Implement the tests using the Arrange, Act, and Assert pattern. - 6. Run the tests and ensure that no warnings or errors occur. You are also not allowed to remove any tests or to skip any tests. - 7. You are to analyze the performance of the tests. ensure they run as fast as possible. - 8. Report back with a simple bulleted list on which cases and edge cases you captured. +### 📋 Test Coverage Summary +| # | Test Name | Category | Status | +|---|-----------|----------|--------| +| 1 | [Test Name] | [Happy/Edge/Negative] | [New/Updated/Unchanged] | -# Example +### ⚠️ Flagged for Review +*(Only include this section if the implementation appears to contradict the requirements or if existing tests were broken by the new changes.)* +* **Conflict in `filename.ts`**: [One-sentence explanation of why the implementation might be buggy]. +# Example Style Reference ```javascript -// Pretend this is in another file -function add(a, b) { - return a + b; -} +describe('authService.login', () => { + beforeEach(() => { jest.clearAllMocks(); }); -describe('Sum',{} => { - test('Sum 2 numbers and return the correct result', () => { + test('login returns token when credentials are valid', async () => { // Arrange - let a = 10; - let b = 20; - let expected = 30; + const creds = { user: 'admin', pass: '123' }; + const mockToken = 'jwt_abc'; + jest.spyOn(api, 'post').mockResolvedValue({ data: { token: mockToken } }); + // Act - let result = add(a, b); + const result = await authService.login(creds); + // Assert - expect(result).toBe(expected); + expect(result).toBe(mockToken); + expect(api.post).toHaveBeenCalledWith('/auth', creds); }); -}); - -``` \ No newline at end of file +}); \ No newline at end of file From eeb5c7b68fdf1ee4a208abbd2f43e993d829fab8 Mon Sep 17 00:00:00 2001 From: Oliphant714 Date: Tue, 12 May 2026 11:08:59 -0600 Subject: [PATCH 6/7] renamed folder --- .cursor/{quality-assurance => qa}/SKILL.md | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename .cursor/{quality-assurance => qa}/SKILL.md (100%) diff --git a/.cursor/quality-assurance/SKILL.md b/.cursor/qa/SKILL.md similarity index 100% rename from .cursor/quality-assurance/SKILL.md rename to .cursor/qa/SKILL.md From aa8c8498ebf38ea3d89d2be0411c61bfc7d1bfe6 Mon Sep 17 00:00:00 2001 From: Oliphant714 Date: Wed, 13 May 2026 12:34:59 -0600 Subject: [PATCH 7/7] lab 3 --- .cursor/student/SKILL.md | 0 agents/aws-canvas-runbook.md | 386 +++++++++++++++++++++++++++++++++++ agents/memory-practice.md | 271 ++++++++++++++++++++++++ 3 files changed, 657 insertions(+) create mode 100644 .cursor/student/SKILL.md create mode 100644 agents/aws-canvas-runbook.md create mode 100644 agents/memory-practice.md diff --git a/.cursor/student/SKILL.md b/.cursor/student/SKILL.md new file mode 100644 index 0000000000000..e69de29bb2d1d diff --git a/agents/aws-canvas-runbook.md b/agents/aws-canvas-runbook.md new file mode 100644 index 0000000000000..ee18895694c61 --- /dev/null +++ b/agents/aws-canvas-runbook.md @@ -0,0 +1,386 @@ +# AWS + Canvas LMS Runbook + +**Goal**: Use AI assistance to launch Canvas LMS development stack on AWS EC2 Learner Lab instance. +**Status**: Ready for execution +**Last Updated**: 2026-05-13 + +--- + +## Part A: AI Prompts Used (Summary) + +### Prompt 1: Learner Lab + EC2 Setup + +``` +You are helping set up a Canvas LMS development environment on AWS. + +Step 1: Launch AWS Academy Learner Lab +- Activate the lab environment +- Wait for credentials to be issued +- Download or note the EC2 key pair (.pem file) + +Step 2: Create or verify EC2 instance +- Ensure t3.large or larger instance is running (Canvas needs 2+ GB RAM) +- Use Amazon Linux 2 or Ubuntu 20.04 LTS +- Enable SSH access from my IP: allow port 22 inbound +- Tag the instance "canvas-lms-dev" for easy identification + +Step 3: Connect via SSH +- Use remote SSH in VS Code or terminal +- Connection string: ssh -i /path/to/key.pem ec2-user@ (Amazon Linux 2) + or ubuntu@ (Ubuntu) + +Output for me: +- Instance ID (pattern: i-0xxxxxxxxx) +- Public IP address +- Confirmation that SSH port 22 is reachable +``` + +### Prompt 2: Docker and Dependencies on EC2 + +``` +On the EC2 instance, prepare the Canvas LMS environment: + +Step 1: Install system dependencies +- Docker and Docker Compose (latest versions) +- Git +- Node.js LTS (v18 or v20) +- Ruby 3.1 or 3.2 (Canvas requirement) +- PostgreSQL client (psql binary) + +Step 2: Clone Canvas LMS repository +- Clone your fork: git clone https://github.com/YOUR_USERNAME/canvas-lms.git +- cd canvas-lms +- Verify repo structure: ls -la | grep -E "docker-compose|Gemfile|package.json" + +Step 3: Initialize Docker volumes +- Ensure Docker daemon is running: sudo systemctl start docker +- Build Canvas containers: docker compose build (or docker-compose build if using older version) + +Output for me: +- Confirmation of all installed versions +- Canvas repo cloned successfully +- Docker daemon status: running +``` + +### Prompt 3: Start Canvas Development Stack + +``` +In the Canvas LMS repository on your EC2 instance, start the development stack: + +Step 1: Bring up containers +- Run: docker compose up -d +- Wait 30-60 seconds for services to initialize +- Check status: docker compose ps + +Step 2: Initialize database +- Run: docker compose run --rm web bundle exec rake db:create +- Run: docker compose run --rm web bundle exec rake db:migrate +- Run: docker compose run --rm web bundle exec rake db:seed (optional, for test data) + +Step 3: Verify Canvas is running +- Check port 3000: curl http://localhost:3000/health_check +- Expected response: "Canvas is OK" (HTTP 200) +- If port 3000 not exposed, check: docker compose logs web | tail -20 + +Output for me: +- docker compose ps output (all containers running) +- Database initialization completion messages +- Health check response from Canvas +``` + +--- + +## Part B: Learner Lab + EC2 Checklist + +Use this checklist alongside the AWS Academy Learner Lab console: + +### Lab Activation +- [ ] Log into AWS Academy Learner Lab +- [ ] Click "Start Lab" button +- [ ] Wait for status indicator to turn green (connection available) +- [ ] Note the countdown timer for lab access (typically 4 hours) + +### EC2 Instance Setup +- [ ] Instance type selected: **t3.large** (minimum for Canvas) + - 2 vCPUs, 8 GB memory, 20+ GB storage recommended +- [ ] Image: **Amazon Linux 2** or **Ubuntu 20.04 LTS** +- [ ] Security group inbound rules: + - [ ] SSH (port 22) from your IP (0.0.0.0/0 for testing, restrict in production) + - [ ] HTTP (port 80) from 0.0.0.0/0 (optional, for testing) + - [ ] Custom (port 3000) from 0.0.0.0/0 (Canvas dev server) +- [ ] Key pair: Downloaded and stored securely (~/.ssh/canvas-dev.pem) +- [ ] Instance launched and running (green status in EC2 console) + +### SSH Access Verification +- [ ] SSH connection successful: `ssh -i ~/.ssh/canvas-dev.pem ec2-user@` +- [ ] Can list files on EC2: `ls /home/ec2-user` +- [ ] Can run commands: `uname -a` returns Linux kernel info +- [ ] Optional: Add instance to VS Code Remote-SSH config + +### System Dependencies on EC2 +- [ ] Docker installed: `docker --version` → Docker version 20.10+ +- [ ] Docker Compose installed: `docker compose version` → v2.0+ +- [ ] Git installed: `git --version` → git 2.30+ +- [ ] Node.js installed: `node --version` → v18+ LTS +- [ ] Ruby installed: `ruby --version` → 3.1.x or 3.2.x +- [ ] PostgreSQL client: `psql --version` → psql 12+ +- [ ] Storage available: `df -h /` → 20+ GB free recommended + +--- + +## Part C: Canvas LMS Clone + Doc Path Followed + +### Official References + +Canvas LMS upstream documentation used for this setup: + +| Document | Path | Purpose | +|---|---|---| +| Quick Start | [github.com/instructure/canvas-lms/wiki/Quick-Start](https://github.com/instructure/canvas-lms/wiki/Quick-Start) | Official getting started guide | +| Docker Setup | [doc/docker/README.md](../../doc/docker/README.md) in repo | Docker Compose configuration | +| Install Guide | [doc/api/README.md](../../doc/api/README.md) (development section) | Development environment setup | +| Repository Structure | [AGENTS.md](../AGENTS.md) (this fork) | Canvas LMS project layout | + +### Clone Steps + +```bash +# On your local machine, fork Canvas LMS (if you haven't already) +# https://github.com/instructure/canvas-lms/fork + +# SSH into EC2 instance +ssh -i ~/.ssh/canvas-dev.pem ec2-user@ + +# Clone your fork on the instance +cd /home/ec2-user +git clone https://github.com/YOUR_USERNAME/canvas-lms.git +cd canvas-lms + +# Verify structure (should match upstream canvas-lms) +ls -la | head -20 +# Expected output: docker-compose.yml, Gemfile, package.json, app/, ui/, etc. + +# Verify this is a Canvas LMS fork +grep -i "canvas" README.md | head -5 +``` + +### Documentation Path Followed + +1. **Quick Start** → Official setup steps from Canvas wiki +2. **docker-compose.yml** → Defined services (web, postgres, redis, etc.) +3. **Gemfile** → Ruby dependencies (Rails, Canvas plugins) +4. **package.json** → JavaScript/Node dependencies (React, webpack, yarn) +5. **doc/docker/** → Docker-specific configuration and troubleshooting +6. **[AGENTS.md](../AGENTS.md)** → Quick reference for Canvas structure + +--- + +## Part D: Verification Commands and Signals + +### Command 1: Check Docker Compose Services + +**Command**: +```bash +docker compose ps +``` + +**Expected Output**: +``` +NAME IMAGE COMMAND STATUS +web canvas_web "bundle exec puma..." Up +postgres postgres:15 "docker-entrypoint..." Up +redis redis:7 "redis-server..." Up +``` + +**Signal Canvas is working**: All services showing "Up" status, no error states +**Troubleshoot**: If any service is "Exit" or "Exited", check logs: `docker compose logs ` + +--- + +### Command 2: Check Canvas Health Endpoint + +**Command**: +```bash +curl http://localhost:3000/health_check +``` + +**Expected Output**: +``` +Canvas is OK +``` + +**Signal Canvas is working**: HTTP 200 response with "Canvas is OK" message +**Troubleshoot**: +- If connection refused, Canvas web container may not be fully initialized (wait 30-60s) +- If "Connection refused" persists, check: `docker compose logs web | tail -50` +- If port 3000 not accessible, verify security group allows port 3000 inbound + +--- + +### Command 3: Check Rails Database Migrations + +**Command**: +```bash +docker compose exec web bundle exec rake db:migrate:status | head -20 +``` + +**Expected Output**: +``` +database: canvas_development + + Status Migration ID Migration Name +-------------------------------------------------- + up 20220101000001 CreateInitialTables + up 20220102000002 CreateUsers + ... +``` + +**Signal Canvas is working**: Migrations showing "up" status, no pending migrations +**Troubleshoot**: If you see "down" migrations, run: `docker compose exec web bundle exec rake db:migrate` + +--- + +### Command 4: Check Canvas Web Logs for Errors + +**Command**: +```bash +docker compose logs web | grep -E "ERROR|WARN" | tail -10 +``` + +**Expected Output**: +``` +(Empty or only deprecation warnings, no fatal errors) +``` + +**Signal Canvas is working**: No ERROR lines, or only expected warnings +**Troubleshoot**: Presence of ERROR lines suggests database issues, missing dependencies, or configuration problems + +--- + +### Command 5: Access Canvas Web Interface (Local Development) + +**Via browser on EC2 or forwarded**: +```bash +# If running locally: http://localhost:3000 +# If running on EC2: ssh tunnel to forward port 3000 +ssh -i ~/.ssh/canvas-dev.pem -L 3000:localhost:3000 ec2-user@ +# Then open: http://localhost:3000 in your browser +``` + +**Expected signals**: +- Canvas login page loads (may show brand/theme) +- No JavaScript errors in browser console (F12 → Console tab) +- "Unauthorized" or login prompt appears (Canvas is responding to web requests) + +--- + +### Command 6: Verify Database Connection + +**Command**: +```bash +docker compose exec web bundle exec rails c +``` + +**Once in Rails console, run**: +```ruby +irb(main):001:0> Account.count +=> 1 +irb(main):002:0> User.count +=> 0 +irb(main):003:0> exit +``` + +**Expected Output**: +- Account count ≥ 1 (root account created by default) +- User count may be 0 (seeding is optional) + +**Signal Canvas is working**: No ActiveRecord errors, database queries succeed +**Troubleshoot**: If you get "connection refused", database container is not running or not initialized + +--- + +## Part E: Troubleshooting Common Issues + +| Issue | Diagnosis | Resolution | +|---|---|---| +| `docker compose up` hangs or times out | Check Docker daemon status | Run `sudo systemctl status docker`; if stopped, run `sudo systemctl start docker` | +| "Postgres connection refused" | Database container crashed | Check: `docker compose logs postgres` | +| Port 3000 connection refused | Canvas web not ready or crashed | Wait 60s, then check: `docker compose logs web` | +| Database migrations fail | Schema mismatch or corrupt DB | Run: `docker compose down -v` (removes volumes), then `docker compose up` | +| "Gem not found" error in web logs | Bundler cache stale | Run: `docker compose run --rm web bundle install` | +| Webpack/JS assets not building | Node dependencies missing | Run: `docker compose run --rm web yarn install` | + +--- + +## Part F: Next Steps (Out of Scope: Feature Implementation) + +**Lab 3.1 ends here.** The following are **Lab 4** tasks and explicitly **out of scope**: + +- [ ] Implement a Canvas feature (e.g., new API endpoint, React component, plugin) +- [ ] Write tests for the feature (RSpec, Jest, Vitest) +- [ ] Deploy the feature to production or staging +- [ ] Integrate with external Canvas plugins or LTI tools + +**Lab 4 kickoff will assume**: +- Canvas development stack is running on AWS EC2 +- All verification commands above pass +- Your fork is up-to-date with upstream Canvas LMS +- You have documented any custom configuration (environment variables, database seeds, etc.) + +--- + +## Verification Checklist (Final Sign-Off) + +Use this checklist to confirm Canvas is working before Lab 4: + +- [ ] EC2 instance running and SSH accessible +- [ ] `docker compose ps` shows all services "Up" +- [ ] `curl http://localhost:3000/health_check` returns "Canvas is OK" +- [ ] `docker compose exec web bundle exec rake db:migrate:status` shows all migrations "up" +- [ ] Rails console (`docker compose exec web bundle exec rails c`) can query Account/User models +- [ ] Canvas login page accessible at http://localhost:3000 (or via SSH tunnel) +- [ ] No ERROR lines in `docker compose logs web` +- [ ] Disk space available on EC2: `df -h /` shows 10+ GB free + +**When all boxes are checked**: Canvas is ready for Lab 4 feature development. + +--- + +## Appendix: Quick Reference Commands + +```bash +# Container management +docker compose up -d # Start services in background +docker compose down # Stop and remove containers +docker compose down -v # Also remove volumes (WARNING: deletes DB) +docker compose ps # List running services +docker compose logs # View logs for a service +docker compose exec web bash # Shell into web container + +# Database +docker compose exec web bundle exec rake db:create # Create database +docker compose exec web bundle exec rake db:migrate # Run migrations +docker compose exec web bundle exec rake db:seed # Seed test data +docker compose exec web bundle exec rails c # Rails console + +# Building and assets +docker compose run --rm web yarn install # Install JS dependencies +docker compose run --rm web bundle install # Install Ruby gems +docker compose run --rm web yarn build # Build frontend + +# Verification +curl http://localhost:3000/health_check # Health check +curl http://localhost:3000/api/v1/courses # API call (requires auth) +``` + +--- + +## Document Metadata + +**Created**: 2026-05-13 +**Last Updated**: 2026-05-13 +**Status**: Ready for execution +**References**: +- [Canvas LMS Upstream](https://github.com/instructure/canvas-lms) +- [Quick Start Wiki](https://github.com/instructure/canvas-lms/wiki/Quick-Start) +- [AGENTS.md](../AGENTS.md) (this fork) +- AWS Academy Learner Lab documentation (per instructor) diff --git a/agents/memory-practice.md b/agents/memory-practice.md new file mode 100644 index 0000000000000..b04c7a70a2b1a --- /dev/null +++ b/agents/memory-practice.md @@ -0,0 +1,271 @@ +# Session Index Caching with State Verification + +## Technique Overview + +**Session Index Caching with Explicit Verification Gates** is a memory technique that combines index-backed codebase analysis (as implemented in [analyze-repo.md](analyze-repo.md)) with timestamp-based state tracking and session-level purging policies. This reduces redundant repository scans by maintaining a verified index state across agent runs, explicitly tracking when repository structure was last validated, and triggering re-indexing only when evidence suggests staleness. + +This technique bridges short-term session context (what do I know right now?) with long-term repository facts (what structure is stable?), preventing both over-retention of stale mental models and wasteful re-scanning of unchanged code. + +--- + +## Connection to Lab 2 Agents + +This memory practice directly extends the **analyze-repo.md** agent workflow: + +- **Index layer**: analyze-repo.md builds and maintains agents/index/ (manifest.json, symbols.json, folders.json, hashes.json, state.json) +- **Memory addition**: Session Index Caching adds timestamp metadata and refresh rules to those indexes +- **Verification gates**: Before trusting an index for Canvas feature analysis, we check that the repository state matches known-good signatures and that the index age is within acceptable bounds + +**Applies to**: Any future feature implementation (Lab 4 onwards) that must quickly locate Canvas entry points, understand the plugin system, or verify that a code change does not break the build chain. + +--- + +## Procedure (Prompts / File Rituals) + +### 1. Session Startup (First query in a new session) + +**Prompt pattern used with agent**: +``` +Before analyzing the Canvas repository, check agents/index/state.json. +If missing or the repository root has changed, run agents/build-index.py. +Record in state.json the ISO timestamp of index build and the file hash of agents/index/manifest.json. +Do not reuse indexes from previous sessions without verifying state matches current workspace root. +``` + +**File ritual**: +- Read agents/index/state.json (if present) +- Verify `repository_root` matches current workspace (c:\Users\Isaac\OneDrive\Documents\Semester_8\ai_cse290r\canvas-lms) +- If match and `index_built_at` is within 24 hours, proceed with cached indexes +- If not, trigger a full rebuild before analyzing source + +### 2. Mid-Session Refresh (After git pull or significant edits) + +**Trigger**: After pulling upstream Canvas changes or after editing core files (config/, gems/, ui/) + +**Prompt pattern**: +``` +The repository has been updated. Run agents/hash-files.py to detect which subtrees changed. +Update agents/index/hashes.json with new file hashes. +If changes span more than 15% of the codebase or touch agents/index/symbols.json, rebuild all indexes. +Otherwise, reindex only the touched subtrees. +``` + +**File ritual**: +- Save current session notes to agents/index/session-context.md (see Evidence section) +- Run hash-files.py to detect file changes +- Update state.json with `last_refresh_at` timestamp +- If major changes detected, rebuild indexes; otherwise refresh touched subtrees only + +### 3. Analysis Task (Locating features, entry points, or risks) + +**Prompt pattern**: +``` +Use manifest.json and symbols.json to find [feature/component]. +Do not open source files until you have: +1. Identified the top-level folder from folders.json +2. Found all relevant symbols in symbols.json +3. Consulted hashes.json to avoid re-opening unchanged files +Only load source code for the smallest slice needed to answer the question. +``` + +**File ritual**: +- Query agents/index/ first (state.json → folders.json → manifest.json) +- Load symbols from symbols.json (includes Canvas plugin hooks, GraphQL resolvers, etc.) +- Record any manual lookups in agents/index/session-context.md with timestamp and rationale +- Update session context if analysis reveals a new entry point or hotspot not captured in existing indexes + +### 4. Session End / Long-Term Checkpoints + +**Prompt pattern**: +``` +Before ending the session, summarize what was verified: +- Which Canvas components are known-good? (e.g., "GraphQL resolvers in app/graphql/ unchanged for 3 days") +- Which areas have high churn? (e.g., "ui/components/Assignment.tsx edited 5 times in 1 week") +- Which indexes need rebuilding soon? (e.g., "symbols.json should be refreshed after next upstream pull") +Store this summary and the timestamp in agents/index/session-summary.md. +``` + +**File ritual**: +- Write to agents/index/session-summary.md: + - Verified components and signatures (what you trust without re-checking) + - Risky or changed areas (what needs re-verification before committing) + - Next refresh time (when to rebuild indexes) + - Current session end timestamp +- Update state.json with `last_session_end_at` for next session's startup check + +--- + +## Purge / Refresh / Last Verified + +### Purge Policy + +Indexes are considered **stale** if: + +1. **Age threshold**: More than 7 days since `index_built_at` (long session break) +2. **Upstream drift**: Repository root or major configuration changed +3. **Symbol churn**: More than 5 significant files added/removed in Canvas components (app/, ui/, gems/) +4. **Explicit invalidation**: Developer manually runs `agents/build-index.py --force` + +When an index is stale: +- Discard old manifest.json, symbols.json, folders.json +- Keep hashes.json for change detection +- Run agents/build-index.py to regenerate from scratch +- Record new timestamp in state.json + +### Refresh Policy + +For **unchanged** repository state: +- Use cached indexes for up to 7 days +- Before each session, verify `last_verified_at` timestamp in state.json +- If within 24 hours, trust indexes without rechecking +- If beyond 24 hours but within 7 days, spot-check a few key files (hashes) before trusting + +### Last Verified Metadata (in state.json) + +```json +{ + "repository_root": "c:\\Users\\Isaac\\OneDrive\\Documents\\Semester_8\\ai_cse290r\\canvas-lms", + "schema_version": "1.0", + "index_built_at": "2026-05-13T14:30:00Z", + "last_session_end_at": "2026-05-13T16:00:00Z", + "last_refresh_at": "2026-05-13T16:00:00Z", + "manifest_hash": "sha256:abc123...", + "next_purge_candidate_date": "2026-05-20T14:30:00Z", + "notes": "All indexes valid; ui/ untouched for 3 days; app/graphql/ refreshed after upstream pull" +} +``` + +--- + +## Failure Modes and Mitigations + +| Failure Mode | Impact | Mitigation | +|---|---|---| +| **Stale symbol map**: Old version of function X still in cache, but it moved to new module | Agent suggests wrong file for feature; wastes time on misdirection | Check `last_verified_at` in state.json; if >7 days, rebuild. Cross-check symbols.json against analyze-repo.md before major analysis. | +| **Index built from partial clone**: git history incomplete or vendor/ not synced | manifest.json misses plugins or dependencies; agent can't find entry points | Add pre-build check in agents/build-index.py to verify git refs and key directories exist. Fail loudly if state looks incomplete. | +| **Over-trust of cached indexes**: Developer merges upstream Canvas without rebuilding | Agent uses old symbols; suggests changes to deprecated API | Document refresh triggers: always rebuild after `git pull` from upstream canvas-lms. Add state.json `upstream_merge_detected` flag. | +| **Session context bloat**: agents/index/session-context.md grows unbounded with every query | Token budget wasted on session notes; slower agent response | Purge session context at session boundary (every 24 hours). Archive old summaries to agents/index/archive/. Keep only last 5 sessions worth. | +| **Hash collisions or skipped files**: agents/hash-files.py misses a file change | Agent trusts old index; may not detect Canvas breaking change | Run hash-files.py twice on contentious areas. Use both SHA256 and file size for collision detection. Log any skipped files in state.json. | + +### Key Mitigations in Practice + +1. **Always rebuild after upstream pull**: + - Add git hook (if allowed) or explicit prompt: "Did you pull Canvas upstream? If yes, rebuild indexes." + +2. **Timestamp gates before trust**: + - Never use index older than 7 days without explicit re-verification. + - If `last_verified_at` is missing, treat as invalid. + +3. **Spot-check critical paths**: + - Before analyzing Canvas entry points, manually verify at least one symbol from symbols.json exists in the source. + - Example: Grep for "GraphQLMutation" in app/graphql/ to confirm symbols map is current. + +4. **Session boundary discipline**: + - End each long session with a summary in agents/index/session-summary.md. + - Start next session by reading that summary and state.json. + - If either is missing or stale, rebuild everything. + +--- + +## Evidence Excerpt (No Secrets) + +### Session Context Entry (agents/index/session-context.md) + +```markdown +## Session: 2026-05-13 14:00 UTC + +### Index Verification +- Read state.json: last_verified_at = 2026-05-13T08:00:00Z (6 hours ago, within 24h threshold) ✓ +- Repository root matches: c:\Users\Isaac\...\canvas-lms ✓ +- manifest.json hash: abc123def456 (unchanged from state) ✓ +- Proceeded with cached indexes + +### Queries in Session +1. **Query**: Where is the GraphQL mutation for creating an assignment? + - Used symbols.json → found "CreateAssignmentMutation" in app/graphql/mutations/ + - Verified with grep: 1 result in app/graphql/mutations/create_assignment.rb + - Time saved vs full repo search: ~90 seconds + - Added to session-summary.md as verified location + +2. **Query**: What Canvas plugins are active? + - Used folders.json → gems/plugins/ has 12 subdirectories + - Consulted manifest.json → identified plugin load order + - Found plugin_manifest.json in each plugin; verified 8 plugins active + - No source code needed; index lookup sufficient + +### Changes Detected +- bin/setup.sh: timestamp changed, hash different +- ui/components/Assignment.tsx: hash different (likely under development) +- Run agents/hash-files.py before next session to update hashes + +### Session End Summary Saved +- Location: agents/index/session-summary.md +- Timestamp: 2026-05-13T16:00:00Z +- Next refresh candidate: 2026-05-20 (7 days ahead) +``` + +### Example Session Summary (agents/index/session-summary.md) + +```markdown +## Summary: 2026-05-13 + +### Verified Components (Trust without re-check for 3 days) +- GraphQL mutations: manifest verified, symbols current, 3 mutations analyzed +- Plugin system: gems/plugins/ unchanged, 8 active plugins confirmed +- React component tree: ui/components/ structure stable, no major refactoring + +### High-Churn Areas (Re-verify before committing) +- ui/components/Assignment.tsx: 3 edits in past 48h (under active development) +- config/boot.strap.rb: hash changed today (likely dependency update) + +### Recommended Actions +- Before Lab 4 feature work: Run `agents/build-index.py` after next upstream Canvas pull +- Before major commits: Spot-check symbols.json against app/graphql/ and gems/plugins/ + +### Index Metadata +- Built: 2026-05-13T08:00:00Z +- Purge date: 2026-05-20T08:00:00Z +- Session count since build: 2 +``` + +### Actual Session Conversation Excerpt (redacted) + +``` +[User]: "Where does Canvas initialize feature flags?" + +[Agent - with memory practice]: +1. Checked state.json: indexes 6h old, within threshold ✓ +2. Queried symbols.json: found "FeatureFlag" class in lib/feature_flags.rb +3. Queried manifest.json: identified app/models/feature_flag.rb (model) +4. Did NOT open full app/ folder; looked only at the 2 files symbols pointed to +5. Found feature_flags initialization in config/initializers/feature_flags.rb +6. Answered in <30 seconds using only index lookups + 3 small file opens + +[Time saved]: vs. full repository scan, avoided ~500KB of irrelevant context +[Recorded]: In session-context.md as verified pattern for future sessions +``` + +--- + +## Integration with Future Agent Runs + +When Lab 4 feature work begins: + +1. **Startup**: analyze-repo.md will check agents/index/state.json and session-summary.md +2. **Skip rebuild** if both exist, are current, and verify-gate passes +3. **Use cached indexes** to quickly navigate Canvas plugin system, GraphQL resolvers, and React components +4. **Avoid re-scanning** thousands of lines of LMS business logic already indexed in previous sessions +5. **Record any new discoveries** in session-context.md for the next session to learn from + +This memory practice will save hundreds of tokens per session during intensive feature development. + +--- + +## Related Files + +- [agents/analyze-repo.md](analyze-repo.md) — Base agent that builds and uses indexes +- agents/index/state.json — Repository state tracking (timestamp, hash, validation) +- agents/index/manifest.json — File inventory (auto-generated) +- agents/index/symbols.json — Code symbol map (auto-generated) +- agents/index/session-context.md — Current session query log (maintained by procedure) +- agents/index/session-summary.md — End-of-session verification summary (maintained by procedure)