diff --git a/Pipfile b/Pipfile index 402f368463..1afca6eb9c 100644 --- a/Pipfile +++ b/Pipfile @@ -52,6 +52,7 @@ evalidate = "==2.1.3" weasyprint = "==68.0" pip = "==26.0" urllib3 = "==2.7.0" +django-storages = {extras = ["s3", "google"], version = "==1.14.6"} [dev-packages] boto3-stubs = { extras = ["s3", "boto3"], version = "==1.43.6" } diff --git a/Pipfile.lock b/Pipfile.lock index de30d92b9a..af26e8990d 100644 --- a/Pipfile.lock +++ b/Pipfile.lock @@ -1,7 +1,7 @@ { "_meta": { "hash": { - "sha256": "d354e34ac5411811aba45a571a092fcfea3bcbd6c2499152b2c94d50aad855b1" + "sha256": "2dc7f6814c129318f6d806f5f00c2fa0fd8c061e04ed23223a70e61c01dcba67" }, "pipfile-spec": 6, "requires": { @@ -222,11 +222,11 @@ }, "asgiref": { "hashes": [ - "sha256:5f184dc43b7e763efe848065441eac62229c9f7b0475f41f80e207a114eda4ce", - "sha256:e8667a091e69529631969fd45dc268fa79b99c92c5fcdda727757e52146ec133" + "sha256:59dcb51c272ad209d59bed5708a64a333083e86017d7fcdd67498eeab7784340", + "sha256:fe386d1c2bff7259ea95929266d12a8cf9a8b5a1c2598402967d8792e7a7c094" ], - "markers": "python_version >= '3.9'", - "version": "==3.11.1" + "markers": "python_version >= '3.10'", + "version": "==3.12.1" }, "attrs": { "hashes": [ @@ -264,11 +264,11 @@ }, "botocore": { "hashes": [ - "sha256:611ad8b1f60661373cd39d9391ff16f1eaf8f5cb1d0a691563a4201d1a2603ce", - "sha256:6257d2655c3abe75eaa49e218b7d883cdc7cea64652b451e5feb08a6c169da3c" + "sha256:0fa62529579469a9224c186eba87e7e561db87dfad20c63ce5d52e427c7fa325", + "sha256:dde324cbe8be14b536b78f4fafe192f94fffcc70716bde804044e62495af8c3c" ], "markers": "python_version >= '3.10'", - "version": "==1.43.8" + "version": "==1.43.66" }, "brotli": { "hashes": [ @@ -386,236 +386,216 @@ }, "certifi": { "hashes": [ - "sha256:3cb2210c8f88ba2318d29b0388d1023c8492ff72ecdde4ebdaddbb13a31b1c4a", - "sha256:8d455352a37b71bf76a79caa83a3d6c25afee4a385d632127b6afb3963f1c580" + "sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775", + "sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55" ], "markers": "python_version >= '3.7'", - "version": "==2026.4.22" + "version": "==2026.7.22" }, "cffi": { "hashes": [ - "sha256:00bdf7acc5f795150faa6957054fbbca2439db2f775ce831222b66f192f03beb", - "sha256:07b271772c100085dd28b74fa0cd81c8fb1a3ba18b21e03d7c27f3436a10606b", - "sha256:087067fa8953339c723661eda6b54bc98c5625757ea62e95eb4898ad5e776e9f", - "sha256:0a1527a803f0a659de1af2e1fd700213caba79377e27e4693648c2923da066f9", - "sha256:0cf2d91ecc3fcc0625c2c530fe004f82c110405f101548512cce44322fa8ac44", - "sha256:0f6084a0ea23d05d20c3edcda20c3d006f9b6f3fefeac38f59262e10cef47ee2", - "sha256:12873ca6cb9b0f0d3a0da705d6086fe911591737a59f28b7936bdfed27c0d47c", - "sha256:19f705ada2530c1167abacb171925dd886168931e0a7b78f5bffcae5c6b5be75", - "sha256:1cd13c99ce269b3ed80b417dcd591415d3372bcac067009b6e0f59c7d4015e65", - "sha256:1e3a615586f05fc4065a8b22b8152f0c1b00cdbc60596d187c2a74f9e3036e4e", - "sha256:1f72fb8906754ac8a2cc3f9f5aaa298070652a0ffae577e0ea9bd480dc3c931a", - "sha256:1fc9ea04857caf665289b7a75923f2c6ed559b8298a1b8c49e59f7dd95c8481e", - "sha256:203a48d1fb583fc7d78a4c6655692963b860a417c0528492a6bc21f1aaefab25", - "sha256:2081580ebb843f759b9f617314a24ed5738c51d2aee65d31e02f6f7a2b97707a", - "sha256:21d1152871b019407d8ac3985f6775c079416c282e431a4da6afe7aefd2bccbe", - "sha256:24b6f81f1983e6df8db3adc38562c83f7d4a0c36162885ec7f7b77c7dcbec97b", - "sha256:256f80b80ca3853f90c21b23ee78cd008713787b1b1e93eae9f3d6a7134abd91", - "sha256:28a3a209b96630bca57cce802da70c266eb08c6e97e5afd61a75611ee6c64592", - "sha256:2c8f814d84194c9ea681642fd164267891702542f028a15fc97d4674b6206187", - "sha256:2de9a304e27f7596cd03d16f1b7c72219bd944e99cc52b84d0145aefb07cbd3c", - "sha256:38100abb9d1b1435bc4cc340bb4489635dc2f0da7456590877030c9b3d40b0c1", - "sha256:3925dd22fa2b7699ed2617149842d2e6adde22b262fcbfada50e3d195e4b3a94", - "sha256:3e17ed538242334bf70832644a32a7aae3d83b57567f9fd60a26257e992b79ba", - "sha256:3e837e369566884707ddaf85fc1744b47575005c0a229de3327f8f9a20f4efeb", - "sha256:3f4d46d8b35698056ec29bca21546e1551a205058ae1a181d871e278b0b28165", - "sha256:44d1b5909021139fe36001ae048dbdde8214afa20200eda0f64c068cac5d5529", - "sha256:45d5e886156860dc35862657e1494b9bae8dfa63bf56796f2fb56e1679fc0bca", - "sha256:4647afc2f90d1ddd33441e5b0e85b16b12ddec4fca55f0d9671fef036ecca27c", - "sha256:4671d9dd5ec934cb9a73e7ee9676f9362aba54f7f34910956b84d727b0d73fb6", - "sha256:53f77cbe57044e88bbd5ed26ac1d0514d2acf0591dd6bb02a3ae37f76811b80c", - "sha256:5eda85d6d1879e692d546a078b44251cdd08dd1cfb98dfb77b670c97cee49ea0", - "sha256:5fed36fccc0612a53f1d4d9a816b50a36702c28a2aa880cb8a122b3466638743", - "sha256:61d028e90346df14fedc3d1e5441df818d095f3b87d286825dfcbd6459b7ef63", - "sha256:66f011380d0e49ed280c789fbd08ff0d40968ee7b665575489afa95c98196ab5", - "sha256:6824f87845e3396029f3820c206e459ccc91760e8fa24422f8b0c3d1731cbec5", - "sha256:6c6c373cfc5c83a975506110d17457138c8c63016b563cc9ed6e056a82f13ce4", - "sha256:6d02d6655b0e54f54c4ef0b94eb6be0607b70853c45ce98bd278dc7de718be5d", - "sha256:6d50360be4546678fc1b79ffe7a66265e28667840010348dd69a314145807a1b", - "sha256:730cacb21e1bdff3ce90babf007d0a0917cc3e6492f336c2f0134101e0944f93", - "sha256:737fe7d37e1a1bffe70bd5754ea763a62a066dc5913ca57e957824b72a85e205", - "sha256:74a03b9698e198d47562765773b4a8309919089150a0bb17d829ad7b44b60d27", - "sha256:7553fb2090d71822f02c629afe6042c299edf91ba1bf94951165613553984512", - "sha256:7a66c7204d8869299919db4d5069a82f1561581af12b11b3c9f48c584eb8743d", - "sha256:7cc09976e8b56f8cebd752f7113ad07752461f48a58cbba644139015ac24954c", - "sha256:81afed14892743bbe14dacb9e36d9e0e504cd204e0b165062c488942b9718037", - "sha256:8941aaadaf67246224cee8c3803777eed332a19d909b47e29c9842ef1e79ac26", - "sha256:89472c9762729b5ae1ad974b777416bfda4ac5642423fa93bd57a09204712322", - "sha256:8ea985900c5c95ce9db1745f7933eeef5d314f0565b27625d9a10ec9881e1bfb", - "sha256:8eca2a813c1cb7ad4fb74d368c2ffbbb4789d377ee5bb8df98373c2cc0dee76c", - "sha256:92b68146a71df78564e4ef48af17551a5ddd142e5190cdf2c5624d0c3ff5b2e8", - "sha256:9332088d75dc3241c702d852d4671613136d90fa6881da7d770a483fd05248b4", - "sha256:94698a9c5f91f9d138526b48fe26a199609544591f859c870d477351dc7b2414", - "sha256:9a67fc9e8eb39039280526379fb3a70023d77caec1852002b4da7e8b270c4dd9", - "sha256:9de40a7b0323d889cf8d23d1ef214f565ab154443c42737dfe52ff82cf857664", - "sha256:a05d0c237b3349096d3981b727493e22147f934b20f6f125a3eba8f994bec4a9", - "sha256:afb8db5439b81cf9c9d0c80404b60c3cc9c3add93e114dcae767f1477cb53775", - "sha256:b18a3ed7d5b3bd8d9ef7a8cb226502c6bf8308df1525e1cc676c3680e7176739", - "sha256:b1e74d11748e7e98e2f426ab176d4ed720a64412b6a15054378afdb71e0f37dc", - "sha256:b21e08af67b8a103c71a250401c78d5e0893beff75e28c53c98f4de42f774062", - "sha256:b4c854ef3adc177950a8dfc81a86f5115d2abd545751a304c5bcf2c2c7283cfe", - "sha256:b882b3df248017dba09d6b16defe9b5c407fe32fc7c65a9c69798e6175601be9", - "sha256:baf5215e0ab74c16e2dd324e8ec067ef59e41125d3eade2b863d294fd5035c92", - "sha256:c649e3a33450ec82378822b3dad03cc228b8f5963c0c12fc3b1e0ab940f768a5", - "sha256:c654de545946e0db659b3400168c9ad31b5d29593291482c43e3564effbcee13", - "sha256:c6638687455baf640e37344fe26d37c404db8b80d037c3d29f58fe8d1c3b194d", - "sha256:c8d3b5532fc71b7a77c09192b4a5a200ea992702734a2e9279a37f2478236f26", - "sha256:cb527a79772e5ef98fb1d700678fe031e353e765d1ca2d409c92263c6d43e09f", - "sha256:cf364028c016c03078a23b503f02058f1814320a56ad535686f90565636a9495", - "sha256:d48a880098c96020b02d5a1f7d9251308510ce8858940e6fa99ece33f610838b", - "sha256:d68b6cef7827e8641e8ef16f4494edda8b36104d79773a334beaa1e3521430f6", - "sha256:d9b29c1f0ae438d5ee9acb31cadee00a58c46cc9c0b2f9038c6b0b3470877a8c", - "sha256:d9b97165e8aed9272a6bb17c01e3cc5871a594a446ebedc996e2397a1c1ea8ef", - "sha256:da68248800ad6320861f129cd9c1bf96ca849a2771a59e0344e88681905916f5", - "sha256:da902562c3e9c550df360bfa53c035b2f241fed6d9aef119048073680ace4a18", - "sha256:dbd5c7a25a7cb98f5ca55d258b103a2054f859a46ae11aaf23134f9cc0d356ad", - "sha256:dd4f05f54a52fb558f1ba9f528228066954fee3ebe629fc1660d874d040ae5a3", - "sha256:de8dad4425a6ca6e4e5e297b27b5c824ecc7581910bf9aee86cb6835e6812aa7", - "sha256:e11e82b744887154b182fd3e7e8512418446501191994dbf9c9fc1f32cc8efd5", - "sha256:e6e73b9e02893c764e7e8d5bb5ce277f1a009cd5243f8228f75f842bf937c534", - "sha256:f73b96c41e3b2adedc34a7356e64c8eb96e03a3782b535e043a986276ce12a49", - "sha256:f93fd8e5c8c0a4aa1f424d6173f14a892044054871c771f8566e4008eaa359d2", - "sha256:fc33c5141b55ed366cfaad382df24fe7dcbc686de5be719b207bb248e3053dc5", - "sha256:fc7de24befaeae77ba923797c7c87834c73648a05a4bde34b3b7e5588973a453", - "sha256:fe562eb1a64e67dd297ccc4f5addea2501664954f2692b69a76449ec7913ecbf" + "sha256:046bfc24911b37851ee1b51aab8bffe713d89c68c6a057b09484ce9fd5f69b4e", + "sha256:06c72bb76605a4b0cd0aad6930b69d4baf7dd5d806cfc409b824191099700e66", + "sha256:0beceaabe56af686895136a2de78db54ecd8e4046b236b8fd6d6cb61389e9bf2", + "sha256:154852545011f779917b11c78db2358d095da62a9a172b78ad0a583ee5adc0d0", + "sha256:194cffa889098ced9976c3fc6340305e43f6303657d298da55366907c05c22d6", + "sha256:19ee6127ee34de7d83ce3d371ebc5ed91addbdcc39f9ab15ce4eb35a4e534971", + "sha256:1a18a57b58cfb21fc28d72e876acf10eaed67a1ed96226f92af4df681d571c4c", + "sha256:1aa5645c30469b09530c4ebca77ebf8f17618293c58f8549cb1a543a50236e7d", + "sha256:1dea0e4d7d4f11f619fe8c1d76caf49e24405b4b5743c0e3be16a500ecd930c9", + "sha256:208f941bb9d18e768138677f0a6d2ce01f590df56043dda1df1535ac57c88517", + "sha256:210019b6c7cf07f081b4c54635c8cf744377001350e29cc0f81c4377b4797735", + "sha256:246fa40ce8645a614ff682e0b70f37134e460eaf93a775e0cbe3cca585a67a80", + "sha256:25792eac27877609e7bb06d42ff88278a6624fff2ba9bbb523c09616b117e80f", + "sha256:27350daa11d4f10c540e6e89dada4c54feb7256ad03e9a4dc075ebad7ba360d1", + "sha256:28907ab9bfb6aa13184cfc17c6b8e1023c5ab6fd7076d8c20a35e59fe04f8f29", + "sha256:2ae64be792b8966f2c69538199728b290e34726562896df1e5dc8ffd8d8188e8", + "sha256:31348097ff5bbe827ccc41795d4dd099d9f0625e7def00ee653c137a490c2a6c", + "sha256:3143d81e29e1e20a9ce10901ec369012947876596f75a222235965f2b7ae832e", + "sha256:3222ba5d678f80a030e6afbcc33dc1ae5cb45facabb61cee2c7016b8432fde48", + "sha256:3311ed60d36f83378794e1009ac6258bafbf81f7888b4caa7b35a521e3f95813", + "sha256:334644fbac4eff73d985a17a91226df55d0f394160c4cfb880e084c8f7161cac", + "sha256:34e261f78cb6ceaaa36f42f2613f4380d94d9c759a9c73c769ee6e0247364632", + "sha256:363e05fa78e15116c3c32c210ee36884fd6b9afa6d440e47112c3bd511d64cb6", + "sha256:398aff33cee2767e3e781d2554c54bd0dff386bb437581e0d8011fde1a942ec1", + "sha256:3d22a20b1fb1632cc72c22f95f7b0d2961c3e1c235f245ba4c606c4771035659", + "sha256:42a494cee34437f05546455144f2b5d9ac09b1face62bcfce597d2e521066688", + "sha256:42e2f76b9455f5a9a844f770bf3e200ed3da0e15f5df3db9c31fe80b04b3d004", + "sha256:42f6930c31dc7f50732c9ae793c2786c7b6b044195967bbdde40bb9be81c4cc0", + "sha256:456a61fa52d579ebf9df2e9552ead5129855dbaff6c1e5a9b1bc408809bdc062", + "sha256:471cee653ae88de62096552e6d24ccb4a5adb8c8c9f10b5054d0122c15bf2779", + "sha256:49cbc70e6542d4ccccb936558d1064a8012541e78f821f955cff24e357776c94", + "sha256:4a7c934f7360e8cd64fe9efadcbd10c7c6364f531e432b9a4bf5ccbc9e0e8b50", + "sha256:4be96343e422f2dfcd12ab5c9f5aebe03f82f737c6bffeca6830b3875cb44aab", + "sha256:4f42141fc14250de6dde5ee7ea4432be017252d91f19c5ad043c084cea629cac", + "sha256:507a24c282e0f42f8ed737cf048572cbf580468da5555764a8331735e9c736b6", + "sha256:51b31d1c98274844cfd7838ce00bfc27c7423a4dc00fc0772fc3331c2cc90676", + "sha256:58acb8ab8e295e6c5ea12f888cbb13cf21511ef2a3303a23f4325c29d17fe5c1", + "sha256:5a59cc1c4442bc3d5c703bf720b51138d0bfc173618807c9ee2490a7541dd3d9", + "sha256:5bb4e7ea95dcd6a014a6fef62e62467d67d8e582326443f3d68e71d6320a9fcf", + "sha256:5c58fe613dc5e5336357eff555824a314d8e43282600435c8d1cb6a7a2fedd13", + "sha256:5e7cecbaadb83884793e05828cee59b210b24583b9c7425d0ba6a754fe22eb4e", + "sha256:616f097f2fe415bc92a247f02e11f634e1f9e9a83d327e3c915c15089c87869e", + "sha256:63bbfd5ded17c4840ac07cd8f1c21ba9d9708141f840b324f422f41b207e3973", + "sha256:64faea20f4e2613363a1a9b9c7dd73058f3ecd00133a511e72ad7c511658f527", + "sha256:661c298b4821edebead0c91edd2b00374d67ad7c5a1f7a91d4442633b79d6a72", + "sha256:68e62fe11f30d5ca8289242866f0a5291402d8529ca2178ab8afc5c9694ae890", + "sha256:6a8dddef476fab96d066d578fc88526767b836ab5ab21754e1d5bf3879c31c7c", + "sha256:6e192623c49c94421616a5778fba35cf0d5a8d000650c1967ef4448ee5cdd990", + "sha256:7225e4514edb64eb6740324353e0da0711954fd8d7da4576755b1c6e09b697cd", + "sha256:75f80557d1389eddbd0de2681f6a390a0c5338c31ddaa821381c203fc3fd50d9", + "sha256:770de9db11e84213beec501cfcaa013b019820ca881e03344dea5844f7876d94", + "sha256:7750c6449dff7864bb9bb27ddfb0267756189201a3afc911d82b3caacd70dfc3", + "sha256:7bde5e4cc5c10140859842b9d383af292b22639a4dffb725314baf45968cef80", + "sha256:7ce713ace7c0e4520535b42b77eaa742c16dab813978064913e5a3cf82973b41", + "sha256:7da0c5eff80f0197f3b3d1232ec5a682a9325f4ae9016a78f5f5ca35f9ced1f5", + "sha256:7dbb61fe3a7699468030f71bbe5f8a0e326a151daa91beb11a6fc1f980c55e1c", + "sha256:811bd1e21d32de12efca32393a0ab3f5133b54fce9bd44b8bd77ab07da14bf6a", + "sha256:8ef53b2de9bcb9197d31854256575d59dbac0cba72ac627bb291ef5eceb74be4", + "sha256:937c0052c05a31ca1daf18de3158eed4dbfcb9cc107adbea227728d647be701e", + "sha256:9d2055050ea716bd38b7f7f1579c275386646b4894c155a3e2f3cd62ed41b7c6", + "sha256:9f8d177621de5cb38ee3e731eda45d421db093ec0739f46a5594babda7987a98", + "sha256:a2d7755bef5a12ed488f4ef1f1b69ee9191d7396083b755a5d2295f6edb4768b", + "sha256:a48d62ab9d6f4f98c983223a547af44be6ca3691074c31cecced6facd3ba2dc1", + "sha256:a4f00aa42f75d6e4595e8866e748cc1705adc0cddfeb2ca86d0d03993d63ba03", + "sha256:a6e721d4b0e45d5b65e87534470e67b18dcd092c83f68fba09f152b9cbc061af", + "sha256:a730a083190634c65cca36ba5f489531576ebd79bcd5c8e172130f6453127231", + "sha256:a931079504ecc49efed7744c476a5c343a92fabf66dec2db95edb1b2fdc770e2", + "sha256:aa9511c62d14da7aacc9b4bf51f3f697a621e83b2d6919008243c3aad168eea3", + "sha256:ab36d55f9ed2d067327667c2fea18dda018eb628dd6347aa01dda6cf1f5d3836", + "sha256:ad2c86c495b899d862ea0f4b42891b8713a3bd45dd4105c7fd51c2a72f39f3a5", + "sha256:aeae0e330c9f6acd681f647d46cefd30c29f93e3392882e792e82080c9691399", + "sha256:b0431303acaea1089ad4b3e9ce4e6518193def1118d4073ca848635ee4ea2e96", + "sha256:b5bdfd1c873d4e093aabc0ca84c4ca6dbc4f752afb5c86f146d9742580c9da2e", + "sha256:baed1e86cc735622097354b9d1281406caf42ff42a886d29faa8e8d1630333be", + "sha256:c1453022f490d2459a11819d83ad1d586e9ff65a12ac3e705ffebd46d3685dcf", + "sha256:c26608d2222fb1e94487e4a387d85f13eb55d5ed725cb25a0c589ac4ee60e7bc", + "sha256:c7659f22557c5a0bc4855cd635f55edec690cc008a40768527762cb9fb263455", + "sha256:c8c69575568085ba0b1b10c0249d779a214aea6f6522e949a0fc9fb0fcb449d0", + "sha256:c8d2c9fd1f2d16f780d15127abb050d13d1a76c03a4bd87d7e4980e45e511e12", + "sha256:ca82be1a1d406ecfe1d25dc16cb33488e5a16bf4438c9fb590484ea29d92478b", + "sha256:cc572dace3f60ef98d7b12ff411d20f5362feb31a0439eab0085bbfd349982d7", + "sha256:d18e5ac0f2f03f4f518d3e23db0f0cad7faa1da8620e9c09461d443bbf6e6692", + "sha256:d28630f5854ab07ab1fd4aba756de52326c82e6be15d414b12793f1975048b54", + "sha256:d9c275eaacd24aa73f94ffd6de08fc3f932424d8b6c376f4bed7cde376fe7bc3", + "sha256:da0e573f9f97159390c89d9f1a9e41908b66d408cc5b58d08cf3847d844c531b", + "sha256:dd31f52ea1086513bb9df30f8fcee9b8918323ae067a3d5b78bc826a000712be", + "sha256:dddad92b554513a31f272570678ba307fb9f618f05e3d4a5eacafff9eae03e1d", + "sha256:df423d40ee8654634421812bc3b196da3f9bd7d32929da813f8394c4348a5358", + "sha256:df913725b79db7bcf03448f36b7bf8815363417d5b58deecf9305e3e30f0f21a", + "sha256:e0bcb7e0f677f543555d2adff3bf19c05f66cdb4796e5ff602442ab2fe3c4ef7", + "sha256:e2d65b31f36619cda3999b78b2aa9632e76b78448e7a56fc4240824200e7c4fc", + "sha256:e6e8cff14d6fb0be70a09c0bdc58096f501952d04624ebf867e0e56da2df8960", + "sha256:f16c709686a78c727bbbf059f92b0bf41c6fc60deec706d2dc19f529175a6125", + "sha256:f24fb43132a4c6b4cb4eb029492919b2db645be6808d738f244fd146c03c32cb", + "sha256:f53e442b08449d42821fa4a4fba000095af9f62742a500f978a9f557ec44339a", + "sha256:f5cfbc5fe74540d335175b656c725d74d90e3730c626d92575eea35029d9afaa", + "sha256:f81b3b8f3d4e343550fa4baa0e479bba9f2d29ce9c2e9b51d1ce1718d7442fcf", + "sha256:f8ec5e643a9a937f64e1999eb9f75d072263751912dc5cd06d3c85f8f44be7c3", + "sha256:fb92203a88b3d3053034db775110081c49d28be6551923805e039924093761e4", + "sha256:fcd22650c908d7b7da162bbfaab594a1227a15d1643a98c68b122ac642fa2264" ], - "markers": "python_version >= '3.9'", - "version": "==2.0.0" + "markers": "python_version >= '3.10'", + "version": "==2.1.1" }, "charset-normalizer": { "hashes": [ - "sha256:007d05ec7321d12a40227aae9e2bc6dca73f3cb21058999a1df9e193555a9dcc", - "sha256:03853ed82eeebbce3c2abfdbc98c96dc205f32a79627688ac9a27370ea61a49c", - "sha256:07d9e39b01743c3717745f4c530a6349eadbfa043c7577eef86c502c15df2c67", - "sha256:08e721811161356f97b4059a9ba7bafb23ea5ee2255402c42881c214e173c6b4", - "sha256:0c96c3b819b5c3e9e165495db84d41914d6894d55181d2d108cc1a69bfc9cce0", - "sha256:0ea948db76d31190bf08bd371623927ee1339d5f2a0b4b1b4a4439a65298703c", - "sha256:0f7eb884681e3938906ed0434f20c63046eacd0111c4ba96f27b76084cd679f5", - "sha256:12a6fff75f6bc66711b73a2f0addfc4c8c15a20e805146a02d147a318962c444", - "sha256:12d8baf840cc7889b37c7c770f478adea7adce3dcb3944d02ec87508e2dcf153", - "sha256:14265bfe1f09498b9d8ec91e9ec9fa52775edf90fcbde092b25f4a33d444fea9", - "sha256:16d971e29578a5e97d7117866d15889a4a07befe0e87e703ed63cd90cb348c01", - "sha256:177a0ba5f0211d488e295aaf82707237e331c24788d8d76c96c5a41594723217", - "sha256:1a87ca9d5df6fe460483d9a5bbf2b18f620cbed41b432e2bddb686228282d10b", - "sha256:1c2a768fdd44ee4a9339a9b0b130049139b8ce3c01d2ce09f67f5a68048d477c", - "sha256:1c2aed2e5e41f24ea8ef1590b8e848a79b56f3a5564a65ceec43c9d692dc7d8a", - "sha256:1dc8b0ea451d6e69735094606991f32867807881400f808a106ee1d963c46a83", - "sha256:1efde3cae86c8c273f1eb3b287be7d8499420cf2fe7585c41d370d3e790054a5", - "sha256:202389074300232baeb53ae2569a60901f7efadd4245cf3a3bf0617d60b439d7", - "sha256:203104ed3e428044fd943bc4bf45fa73c0730391f9621e37fe39ecf477b128cb", - "sha256:2257141f39fe65a3fdf38aeccae4b953e5f3b3324f4ff0daf9f15b8518666a2c", - "sha256:298930cec56029e05497a76988377cbd7457ba864beeea92ad7e844fe74cd1f1", - "sha256:2cd4a60d0e2fb04537162c62bbbb4182f53541fe0ede35cdf270a1c1e723cc42", - "sha256:2d6eb928e13016cea4f1f21d1e10c1cebd5a421bc57ddf5b1142ae3f86824fab", - "sha256:2fe249cb4651fd12605b7288b24751d8bfd46d35f12a20b1ba33dea122e690df", - "sha256:30b8d1d8c52a48c2c5690e152c169b673487a2a58de1ec7393196753063fcd5e", - "sha256:320ade88cfb846b8cd6b4ddf5ee9e80ee0c1f52401f2456b84ae1ae6a1a5f207", - "sha256:3534e7dcbdcf757da6b85a0bbf5b6868786d5982dd959b065e65481644817a18", - "sha256:36836d6ff945a00b88ba1e4572d721e60b5b8c98c155d465f56ad19d68f23734", - "sha256:38c0109396c4cfc574d502df99742a45c72c08eff0a36158b6f04000043dbf38", - "sha256:3946fa46a0cf3e4c8cb1cc52f56bb536310d34f25f01ca9b6c16afa767dab110", - "sha256:3bec022aec2c514d9cf199522a802bd007cd588ab17ab2525f20f9c34d067c18", - "sha256:3c9a494bc5ec77d43cea229c4f6db1e4d8fe7e1bbffa8b6f0f0032430ff8ab44", - "sha256:3dce51d0f5e7951f8bb4900c257dad282f49190fdbebecd4ba99bcc41fef404d", - "sha256:3dedcc22d73ec993f42055eff4fcfed9318d1eeb9a6606c55892a26964964e48", - "sha256:4042d5c8f957e15221d423ba781e85d553722fc4113f523f2feb7b188cc34c5e", - "sha256:481551899c856c704d58119b5025793fa6730adda3571971af568f66d2424bb5", - "sha256:4dc1e73c36828f982bfe79fadf5919923f8a6f4df2860804db9a98c48824ce8d", - "sha256:4e5163c14bffd570ef2affbfdd77bba66383890797df43dc8b4cc7d6f500bf53", - "sha256:511ef87c8aec0783e08ac18565a16d435372bc1ac25a91e6ac7f5ef2b0bff790", - "sha256:532bc9bf33a68613fd7d65e4b1c71a6a38d7d42604ecf239c77392e9b4e8998c", - "sha256:54523e136b8948060c0fa0bc7b1b50c32c186f2fceee897a495406bb6e311d2b", - "sha256:5649fd1c7bade02f320a462fdefd0b4bd3ce036065836d4f42e0de958038e116", - "sha256:56be790f86bfb2c98fb742ce566dfb4816e5a83384616ab59c49e0604d49c51d", - "sha256:5b77459df20e08151cd6f8b9ef8ef1f961ef73d85c21a555c7eed5b79410ec10", - "sha256:5ed6ab538499c8644b8a3e18debabcd7ce684f3fa91cf867521a7a0279cab2d6", - "sha256:6178f72c5508bfc5fd446a5905e698c6212932f25bcdd4b47a757a50605a90e2", - "sha256:6370e8686f662e6a3941ee48ed4742317cafbe5707e36406e9df792cdb535776", - "sha256:64f02c6841d7d83f832cd97ccf8eb8a906d06eb95d5276069175c696b024b60a", - "sha256:65bcd23054beab4d166035cabbc868a09c1a49d1efe458fe8e4361215df40265", - "sha256:66671f93accb62ed07da56613636f3641f1a12c13046ce91ffc923721f23c008", - "sha256:6696b7688f54f5af4462118f0bfa7c1621eeb87154f77fa04b9295ce7a8f2943", - "sha256:6785f414ae0f3c733c437e0f3929197934f526d19dfaa75e18fdb4f94c6fb374", - "sha256:67f6279d125ca0046a7fd386d01b311c6363844deac3e5b069b514ba3e63c246", - "sha256:6c114670c45346afedc0d947faf3c7f701051d2518b943679c8ff88befe14f8e", - "sha256:6e0d51f618228538a3e8f46bd246f87a6cd030565e015803691603f55e12afb5", - "sha256:6ed74185b2db44f41ef35fd1617c5888e59792da9bbc9190d6c7300617182616", - "sha256:708838739abf24b2ceb208d0e22403dd018faeef86ddac04319a62ae884c4f15", - "sha256:715479b9a2802ecac752a3b0efa2b0b60285cf962ee38414211abdfccc233b41", - "sha256:733784b6d6def852c814bce5f318d25da2ee65dd4839a0718641c696e09a2960", - "sha256:750e02e074872a3fad7f233b47734166440af3cdea0add3e95163110816d6752", - "sha256:752a45dc4a6934060b3b0dab47e04edc3326575f82be64bc4fc293914566503e", - "sha256:7579e913a5339fb8fa133f6bbcfd8e6749696206cf05acdbdca71a1b436d8e72", - "sha256:7641bb8895e77f921102f72833904dcd9901df5d6d72a2ab8f31d04b7e51e4e7", - "sha256:7804338df6fcc08105c7745f1502ba68d900f45fd770d5bdd5288ddccb8a42d8", - "sha256:80d04837f55fc81da168b98de4f4b797ef007fc8a79ab71c6ec9bc4dd662b15b", - "sha256:813c0e0132266c08eb87469a642cb30aaff57c5f426255419572aaeceeaa7bf4", - "sha256:82b271f5137d07749f7bf32f70b17ab6eaabedd297e75dce75081a24f76eb545", - "sha256:84c018e49c3bf790f9c2771c45e9313a08c2c2a6342b162cd650258b57817706", - "sha256:8751d2787c9131302398b11e6c8068053dcb55d5a8964e114b6e196cf16cb366", - "sha256:8778f0c7a52e56f75d12dae53ae320fae900a8b9b4164b981b9c5ce059cd1fcb", - "sha256:87fad7d9ba98c86bcb41b2dc8dbb326619be2562af1f8ff50776a39e55721c5a", - "sha256:8d828b6667a32a728a1ad1d93957cdf37489c57b97ae6c4de2860fa749b8fc1e", - "sha256:8e385e4267ab76874ae30db04c627faaaf0b509e1ccc11a95b3fc3e83f855c00", - "sha256:92a0a01ead5e668468e952e4238cccd7c537364eb7d851ab144ab6627dbbe12f", - "sha256:94e1885b270625a9a828c9793b4d52a64445299baa1fea5a173bf1d3dd9a1a5a", - "sha256:a180c5e59792af262bf263b21a3c49353f25945d8d9f70628e73de370d55e1e1", - "sha256:a277ab8928b9f299723bc1a2dabb1265911b1a76341f90a510368ca44ad9ab66", - "sha256:a5fe03b42827c13cdccd08e6c0247b6a6d4b5e3cdc53fd1749f5896adcdc2356", - "sha256:a6c5863edfbe888d9eff9c8b8087354e27618d9da76425c119293f11712a6319", - "sha256:a89c23ef8d2c6b27fd200a42aa4ac72786e7c60d40efdc76e6011260b6e949c4", - "sha256:adb2597b428735679446b46c8badf467b4ca5f5056aae4d51a19f9570301b1ad", - "sha256:ae196f021b5e7c78e918242d217db021ed2a6ace2bc6ae94c0fc596221c7f58d", - "sha256:ae89db9e5f98a11a4bf50407d4363e7b09b31e55bc117b4f7d80aab97ba009e5", - "sha256:aed52fea0513bac0ccde438c188c8a471c4e0f457c2dd20cdbf6ea7a450046c7", - "sha256:aef65cd602a6d0e0ff6f9930fcb1c8fec60dd2cfcb6facaf4bdb0e5873042db0", - "sha256:af21eb4409a119e365397b2adbaca4c9ccab56543a65d5dbd9f920d6ac29f686", - "sha256:b14b2d9dac08e28bb8046a1a0434b1750eb221c8f5b87a68f4fa11a6f97b5e34", - "sha256:bb6d88045545b26da47aa879dd4a89a71d1dce0f0e549b1abcb31dfe4a8eac49", - "sha256:bb8cc7534f51d9a017b93e3e85b260924f909601c3df002bcdb58ddb4dc41a5c", - "sha256:bc17a677b21b3502a21f66a8cc64f5bfad4df8a0b8434d661666f8ce90ac3af1", - "sha256:bd6c2a1c7573c64738d716488d2cdd3c00e340e4835707d8fdb8dc1a66ef164e", - "sha256:bd9b23791fe793e4968dba0c447e12f78e425c59fc0e3b97f6450f4781f3ee60", - "sha256:c03a41a8784091e67a39648f70c5f97b5b6a37f216896d44d2cdcb82615339a0", - "sha256:c0f081d69a6e58272819b70288d3221a6ee64b98df852631c80f293514d3b274", - "sha256:c35abb8bfff0185efac5878da64c45dafd2b37fb0383add1be155a763c1f083d", - "sha256:c36c333c39be2dbca264d7803333c896ab8fa7d4d6f0ab7edb7dfd7aea6e98c0", - "sha256:c45e9440fb78f8ddabcf714b68f936737a121355bf59f3907f4e17721b9d1aae", - "sha256:c593052c465475e64bbfe5dbd81680f64a67fdc752c56d7a0ae205dc8aeefe0f", - "sha256:cdd68a1fb318e290a2077696b7eb7a21a49163c455979c639bf5a5dcdc46617d", - "sha256:ce3412fbe1e31eb81ea42f4169ed94861c56e643189e1e75f0041f3fe7020abe", - "sha256:cf1493cd8607bec4d8a7b9b004e699fcf8f9103a9284cc94962cb73d20f9d4a3", - "sha256:cf29836da5119f3c8a8a70667b0ef5fdca3bb12f80fd06487cfa575b3909b393", - "sha256:d4a48e5b3c2a489fae013b7589308a40146ee081f6f509e047e0e096084ceca1", - "sha256:d560742f3c0d62afaccf9f41fe485ed69bd7661a241f86a3ef0f0fb8b1a397af", - "sha256:d6038d37043bced98a66e68d3aa2b6a35505dc01328cd65217cefe82f25def44", - "sha256:d61f00a0869d77422d9b2aba989e2d24afa6ffd552af442e0e58de4f35ea6d00", - "sha256:d635aab80466bc95771bb78d5370e74d36d1fe31467b6b29b8b57b2a3cd7d22c", - "sha256:dca4bbc466a95ba9c0234ef56d7dd9509f63da22274589ebd4ed7f1f4d4c54e3", - "sha256:dd915403e231e6b1809fe9b6d9fc55cf8fb5e02765ac625d9cd623342a7905d7", - "sha256:e044c39e41b92c845bc815e5ae4230804e8e7bc29e399b0437d64222d92809dd", - "sha256:e060d01aec0a910bdccb8be71faf34e7799ce36950f8294c8bf612cba65a2c9e", - "sha256:e1421b502d83040e6d7fb2fb18dff63957f720da3d77b2fbd3187ceb63755d7b", - "sha256:e17b8d5d6a8c47c85e68ca8379def1303fd360c3e22093a807cd34a71cd082b8", - "sha256:e5f4d355f0a2b1a31bc3edec6795b46324349c9cb25eed068049e4f472fb4259", - "sha256:e712b419df8ba5e42b226c510472b37bd57b38e897d3eca5e8cfd410a29fa859", - "sha256:e74327fb75de8986940def6e8dee4f127cc9752bee7355bb323cc5b2659b6d46", - "sha256:e80c8378d8f3d83cd3164da1ad2df9e37a666cdde7b1cb2298ed0b558064be30", - "sha256:e8ac484bf18ce6975760921bb6148041faa8fef0547200386ea0b52b5d27bf7b", - "sha256:eca9705049ad3c7345d574e3510665cb2cf844c2f2dcfe675332677f081cbd46", - "sha256:ed065083d0898c9d5b4bbec7b026fd755ff7454e6e8b73a67f8c744b13986e24", - "sha256:edac0f1ab77644605be2cbba52e6b7f630731fc42b34cb0f634be1a6eface56a", - "sha256:effc3f449787117233702311a1b7d8f59cba9ced946ba727bdc329ec69028e24", - "sha256:f22dec1690b584cea26fade98b2435c132c1b5f68e39f5a0b7627cd7ae31f1dc", - "sha256:f495a1652cf3fbab2eb0639776dad966c2fb874d79d87ca07f9d5f059b8bd215", - "sha256:f496c9c3cc02230093d8330875c4c3cdfc3b73612a5fd921c65d39cbcef08063", - "sha256:f59099f9b66f0d7145115e6f80dd8b1d847176df89b234a5a6b3f00437aa0832", - "sha256:f59ad4c0e8f6bba240a9bb85504faa1ab438237199d4cce5f622761507b8f6a6", - "sha256:fbccdc05410c9ee21bbf16a35f4c1d16123dcdeb8a1d38f33654fa21d0234f79", - "sha256:fea24543955a6a729c45a73fe90e08c743f0b3334bbf3201e6c4bc1b0c7fa464" + "sha256:0327fcd59a935777d83410750c50600ee9571af2846f71ce40f25b13da1ef380", + "sha256:03d07803992c6c7bbc976327f34b18b6160327fc81cb82c9d504720ac0be3b62", + "sha256:04ce310cb89c15df659582aee80a0603788732a5e017d5bd5c81158106ce249c", + "sha256:0d861473f743244d349b50f850d10eb87aeb22bbdcc8e64f79273c94af5a8226", + "sha256:0e94703ec9684807f20cfb5eed95c70f67f2a8f21ad620146d7b5a13677b93e5", + "sha256:0fa1aec2d32bcc03c8fa0f6f1712caad1adc38509f31142112e5c9daf5b9c833", + "sha256:16b65ea0f2465b6fb52aa22de5eca612aa964ddfec00a912e26f4656cbef890b", + "sha256:16d10d789dd9bcca1173c95af82c58433122564b7bc39385124be735a35cbe99", + "sha256:19ac87f93086ce37b86e098888555c4b4bc48102279bae3350098c0ed664b501", + "sha256:1d22856ffbe153a602df38e4a5464f0b748a54002e0d69ac6d2ad0a197cc99ec", + "sha256:21e764fd1e70b6a3e205a0e46f3051701f98a8cb3fad66eeb80e48bb502f8698", + "sha256:231ddcbb35e2ff8973e1365db41fe0572662893b99a05deb183b68ad4c0c8bd4", + "sha256:253a4a220747e8b5faf57ec320c4f5efb0cef05f647420bf267143ec15dba10a", + "sha256:280081916dc341820640489a66e4696049401ef1cf6dd672f672e70ad915aca3", + "sha256:2a441ea71902098ffe78c5abe6c494f44160b4af614ed16c3d9a3b1d17fd8ee2", + "sha256:304b13570067b2547562e308af560b3963857b1fa90bd6afd978130130fe2d6a", + "sha256:32286a2c8d167e897177b673176c1e3e00d4057caf5d2b64eef9a3666b03018e", + "sha256:33bdcc2a32c0a0e861f60841a512c8acc658c87c2ac59d89e3a46dacf7d866e4", + "sha256:375b83ed0aecfce76c16d198fbc21f3b11b337d68662bea0a995046682a11419", + "sha256:3c09a49d6cde137258beb3d551994a2927fd35ad5cf96aed573f61bbd67c5f84", + "sha256:3d92613ec25e43b05f042302531ec0f00b8445190e43325880cbd6ab7c2581da", + "sha256:40a126142a56b2dfc0aacbad1de8310cbf60da7656db0e6b16eebd48e3e93519", + "sha256:416c229f77e5ea25b3dfd4b582f8d73d7e43c22320302b9ab128a2d3a0b38efe", + "sha256:432786d3561e69aeeae6c7e8648964ce0ad05736120135601f87ac26b9c83381", + "sha256:43b9e366a31fdd1c87d0eb08f579b4a82b723ea54338f040d6b4e518a026ea29", + "sha256:440eede837960000d74978f0eba527be106b5b9aee0daf779d395276ed0b0614", + "sha256:45b0cc4e3556cd875e09102988d1ab8356c998b596c9fced84547c8138b487a0", + "sha256:476743fe6dfe14a2da12e3ac79125dc84a3b2cf8094369a47a1529b0cd8549fe", + "sha256:4773092f8019072343a7447203308b176e10199920eb02d6195e81bbb3274c29", + "sha256:4b3dac63058cc36820b0dd072f89898604e2d39686fe05321729d00d8ac185a0", + "sha256:4d1c96a7a18b9690a4d46df09e3e3382406ae3213727cd1019ebade1c4a81917", + "sha256:51307f5c71007673a2bf8232ad973483d281e74cb99c8c5a990af1eefa6277d9", + "sha256:51447e9aa2684679af07ca5021c3db526e0284347ebf4ffcec1154c3350cfe32", + "sha256:58150c9f9b9a552505912d182ccdf26f6396fb6094816ceebcbb20eecabaed94", + "sha256:5b10cd92fc5c498b35a8635df6d5a100207f88b63a4dc1de7ef9a548e1e2cd63", + "sha256:5e226f6218febc71f6c1fc2fafb91c226f75bdc1d8fb12d66823716e891608fd", + "sha256:609b3ba8fcc0fb5ab7af00719d0fb6ad0cb518e48e7712d12fd68f1327951198", + "sha256:60f44ade2cf573dad7a277e6f8ca9a51a21dda572b13bd7d8539bb3cd5dbedde", + "sha256:611057cc5d5c0afc743ba8be6bd828c17e0aaa8643f9d0a9b9bb7dea80eb8012", + "sha256:6366a16e1a25018694d6a5d784d09b046edc9eac40ea2b54065c3052672516a1", + "sha256:65a7ff3f705e57d392f7261b6d0550fe137c3019477431f1c355e0db0a7d3e15", + "sha256:673611bbd43f0810bec0b0f028ddeaaa501190339cac411f347ac76917c3ae7b", + "sha256:67830fc78e67501f47bb950471b2dcb9b35b140084429318e862895a8e89c993", + "sha256:68ce9f4d6b26d5ccbf7fd4459bf75f74a0a146677ebba80597df60cbdb20e6f4", + "sha256:68e5f26a1ad57ded6d1cfb85331d1c1a195314756471d97758c48498bb4dcdf5", + "sha256:69b157c5d3292bcd443faca052f3096f637f1e074b98212a933c074ae23dc3b8", + "sha256:75286256590a6320cf106a0d28970d3560aad9ee09aa7b34fb40524792436d35", + "sha256:78841cccf1af7b40f6f716338d50c0902dbe88d9f800b3c973b7a9a0a693a642", + "sha256:78fa18e436a1a0e58dbd7e02fc4473f3f32cceb12df9dfca542d075961c307d2", + "sha256:79580094b00d1789d1f93ea55bc43cb2f611910c72235b7657f3482ddcc1b22d", + "sha256:7b86a2b16095d250c6f58b3d9b2eee6f4147754344f3dab0922f7c9bf7d226c9", + "sha256:83aed2c10721ddd90f68140685391b50811a880af20654c59af6b6c66c40513c", + "sha256:84fd18bcc17526fc2b3c1af7d2b9217d32c9c04448c16ec693b9b4f1985c3d33", + "sha256:871ff67ea1aad4dfd91736464934d56b32dac49f9fbe16cddba36198a7b3a0db", + "sha256:898f0e9068ca27d37f8e83a5b962821df851532e6c4a7d615c1c033f9da6eedf", + "sha256:8a79d9f4d8001473a30c163556b3c3bfebec837495a412dde78b51672f6134f9", + "sha256:8c041122946b7ba21bb32c45b1aa57b1be35527690aeb3c5c234521085632eee", + "sha256:90c44bc373b7687f6948b693cceaea1348ae0975d7474746559494468e3c1d84", + "sha256:9104ed0bd76a429d46f9ec0dbc9b08ad1d2dcdf2b00a5a0daa1c145329b35b44", + "sha256:920079c3f7456fa213e0829ed2073aaa727fd39d889ead5b4f35d0de5460d04f", + "sha256:93d59d504b230e83c7a843251681959a0b6a9cd76f6e146ce1b8a80eb8739af9", + "sha256:9b2aff1c7b3884512b9512c3eaadd9bab39fb45042ffaaa1dd08ff2b9f8109d9", + "sha256:9b8e0f3107e2200b76f6054de99016eac3ee6762713587b36baaa7e4bd2ae177", + "sha256:9bb41182d93ea91f60b4bc8fbf4c820c69ef8a12ab2d917f3f1834f1acad07e8", + "sha256:9cdef90ae47919cae358d8ab15797a800ed41da7aba5d72419fb510729e2ed4b", + "sha256:a1786910334ed46ab1dd73222f2cd1e05c2c3bb39f6dddb4f8b36fc382058a39", + "sha256:a4cfde78a9f2880208d16a93b795726a3017d5977e08d1e162a7a31322479c41", + "sha256:a4fbdde9dd4a9ce5fd52c2b3a347bb50cc89483ef783f1cb00d408c13f7a96c0", + "sha256:aa99adc8f081b475a12843953db36831eaf83ec33eb46a90629ca6a5de45a616", + "sha256:ac351b3b8014eead140e77e9717e2992c6bbe30b63bc3422422eb84865412e3d", + "sha256:ad41ba96094304aa090f5a30cb6e4fb3b3f1c264c523394b4c39bbacc4dc92ba", + "sha256:b5314963fce9b0b12743891de876e724997864ee22aa496f903f426c7e2fa5b2", + "sha256:bcf74c1df76758a395bf0af608c04c82257523f55c9868b334f06270d0f2112b", + "sha256:bd47ba7fc3ca94896759ea0109775132d3e7ab921fbf54038e1bab2e46c313c9", + "sha256:c0323c9daef75ef2e5083624b4585018a0c9d5e3b40f607eed81a311270b934b", + "sha256:c1225416b463483160e4af85d5fc3a9690ccb53fd4b1865a6437825f5ede3209", + "sha256:c1c948747b03be832dceed96ca815cef7360de9aa19d37c730f8e3f6101aca48", + "sha256:c25fe15c70c59eb7c5ce8c06a1f3fa1da0ecc5ea1e7a5922c40fd2fa9b0d5046", + "sha256:cc1b0fff8ead343dae06305f954eb8468ba0ec1a97881f42489d198e4ce3c632", + "sha256:cd6280cf040f233bd7d3407b743b4b4c74f70e8e1c4199cb112a62c941c0772a", + "sha256:cd6c3d4b783c556fa00bf540854e42f135e2f256abd29669fcd0da0f2dec79c2", + "sha256:d4d6fcde76f94f5cb9e43e9e9a61f16dacefd228cbbf6f1a09bd9b219a92f1a1", + "sha256:ddf4af30b417d9fe16481e9b81c27ab2a7cde1ff7ba3e85653b02db7d145dc7b", + "sha256:df115d4d83168fdf2cae48ef1ff6d1cb4c466364e30861b37121de0f3bf1b990", + "sha256:df7276909358e5635ae203673ab7e509ddd224225a8d6b0790bf13eb2bde1cc5", + "sha256:e4fd89cc178bced6ad29cb3e6dd4aa63fa5017c3524dbd0b25998fb64a87cc8b", + "sha256:e9701d0049d92c16703a42771b98d560b95248949f23f8cf7b4eddd201814fb9", + "sha256:ee2f2a527e3c1a6e6411eb4209642e138b544a2d72fe5d0d76daf77b24063534", + "sha256:f7fb7d750cfa0a070d2c24e831fd3481019a60dd317ea2b39acbcebc08b6ed81", + "sha256:f840ed6d8ecba8255df8c42b87fadeda98ddfc6eeec05e2dc66e26d46dd6f58a", + "sha256:f86c6358749bd4fda175388691e3ba8c46e24c5347d0afd20f9b7edfc9faf07d", + "sha256:fa36ec09ef71d158186bc79e359ff5fdd6e7996fe8ab638f00d6b93139ba4fcf", + "sha256:fe2c7201c642b7c308f1675355ad7ff7b66acfe3541625efe5a3ad38f29d6115" ], "markers": "python_version >= '3.7'", - "version": "==3.4.7" + "version": "==3.4.9" }, "click": { "hashes": [ @@ -650,58 +630,55 @@ }, "cryptography": { "hashes": [ - "sha256:0890f502ddf7d9c6426129c3f49f5c0a39278ed7cd6322c8755ffca6ee675a13", - "sha256:0c558d2cdffd8f4bbb30fc7134c74d2ca9a476f830bb053074498fbc86f41ed6", - "sha256:16cd65b9330583e4619939b3a3843eec1e6e789744bb01e7c7e2e62e33c239c8", - "sha256:18349bbc56f4743c8b12dc32e2bccb2cf83ee8b69a3bba74ef8ae857e26b3d25", - "sha256:1e2d54c8be6152856a36f0882ab231e70f8ec7f14e93cf87db8a2ed056bf160c", - "sha256:22a5cb272895dce158b2cacdfdc3debd299019659f42947dbdac6f32d68fe832", - "sha256:27241b1dc9962e056062a8eef1991d02c3a24569c95975bd2322a8a52c6e5e12", - "sha256:2b4d59804e8408e2fea7d1fbaf218e5ec984325221db76e6a241a9abd6cdd95c", - "sha256:2eb992bbd4661238c5a397594c83f5b4dc2bc5b848c365c8f991b6780efcc5c7", - "sha256:369a6348999f94bbd53435c894377b20ab95f25a9065c283570e70150d8abc3c", - "sha256:3cb07a3ed6431663cd321ea8a000a1314c74211f823e4177fefa2255e057d1ec", - "sha256:40ba1f85eaa6959837b1d51c9767e230e14612eea4ef110ee8854ada22da1bf5", - "sha256:4defde8685ae324a9eb9d818717e93b4638ef67070ac9bc15b8ca85f63048355", - "sha256:55b7718303bf06a5753dcdccf2f3945cf18ad7bffde41b61226e4db31ab89a9c", - "sha256:561215ea3879cb1cbbf272867e2efda62476f240fb58c64de6b393ae19246741", - "sha256:58d00498e8933e4a194f3076aee1b4a97dfec1a6da444535755822fe5d8b0b86", - "sha256:59baa2cb386c4f0b9905bd6eb4c2a79a69a128408fd31d32ca4d7102d4156321", - "sha256:5a5ed8fde7a1d09376ca0b40e68cd59c69fe23b1f9768bd5824f54681626032a", - "sha256:5b012212e08b8dd5edc78ef54da83dd9892fd9105323b3993eff6bea65dc21d7", - "sha256:5c3932f4436d1cccb036cb0eaef46e6e2db91035166f1ad6505c3c9d5a635920", - "sha256:614d0949f4790582d2cc25553abd09dd723025f0c0e7c67376a1d77196743d6e", - "sha256:76341972e1eff8b4bea859f09c0d3e64b96ce931b084f9b9b7db8ef364c30eff", - "sha256:77a2ccbbe917f6710e05ba9adaa25fb5075620bf3ea6fb751997875aff4ae4bd", - "sha256:7995ef305d7165c3f11ae07f2517e5a4f1d5c18da1376a0a9ed496336b69e5f3", - "sha256:7ce4bfae76319a532a2dc68f82cc32f5676ee792a983187dac07183690e5c66f", - "sha256:7e8eac43dfca5c4cccc6dad9a80504436fca53bb9bc3100a2386d730fbe6b602", - "sha256:84cf79f0dc8b36ac5da873481716e87aef31fcfa0444f9e1d8b4b2cece142855", - "sha256:8c7378637d7d88016fa6791c159f698b3d3eed28ebf844ac36b9dc04a14dae18", - "sha256:8cd666227ef7af430aa5914a9910e0ddd703e75f039cef0825cd0da71b6b711a", - "sha256:906cbf0670286c6e0044156bc7d4af9cbb0ef6db9f73e52c3ec56ba6bdde5336", - "sha256:9071196d81abc88b3516ac8cdfad32e2b66dd4a5393a8e68a961e9161ddc6239", - "sha256:9249e3cd978541d665967ac2cb2787fd6a62bddf1e75b3e347a594d7dacf4f74", - "sha256:984a20b0f62a26f48a3396c72e4bc34c66e356d356bf370053066b3b6d54634a", - "sha256:9be5aafa5736574f8f15f262adc81b2a9869e2cfe9014d52a44633905b40d52c", - "sha256:9c459db21422be75e2809370b829a87eb37f74cd785fc4aa9ea1e5f43b47cda4", - "sha256:9ccdac7d40688ecb5a3b4a604b8a88c8002e3442d6c60aead1db2a89a041560c", - "sha256:a0e692c683f4df67815a2d258b324e66f4738bd7a96a218c826dce4f4bd05d8f", - "sha256:a5da777e32ffed6f85a7b2b3f7c5cbc88c146bfcd0a1d7baf5fcc6c52ee35dd4", - "sha256:a64697c641c7b1b2178e573cbc31c7c6684cd56883a478d75143dbb7118036db", - "sha256:ad64688338ed4bc1a6618076ba75fd7194a5f1797ac60b47afe926285adb3166", - "sha256:bd72e68b06bb1e96913f97dd4901119bc17f39d4586a5adf2d3e47bc2b9d58b5", - "sha256:c17dfe85494deaeddc5ce251aebd1d60bbe6afc8b62071bb0b469431a000124f", - "sha256:c18684a7f0cc9a3cb60328f496b8e3372def7c5d2df39ac267878b05565aaaae", - "sha256:cc90c0b39b2e3c65ef52c804b72e3c58f8a04ab2a1871272798e5f9572c17d20", - "sha256:db63bf618e5dea46c07de12e900fe1cdd2541e6dc9dbae772a70b7d4d4765f6a", - "sha256:ea8990436d914540a40ab24b6a77c0969695ed52f4a4874c5137ccf7045a7057", - "sha256:ecde28a596bead48b0cfd2a1b4416c3d43074c2d785e3a398d7ec1fc4d0f7fbb", - "sha256:f5333311663ea94f75dd408665686aaf426563556bb5283554a3539177e03b8c", - "sha256:fdfef35d751d510fcef5252703621574364fec16418c4a1e5e1055248401054b" + "sha256:031e2d5dd4bb9caa3ca9c82e5a197fd8ae680232cee62603d1a813f3f07e3d03", + "sha256:06a32a980526a6ab9a4b9bf8f7385800791e2bb960903cb6b530e4817509a3b7", + "sha256:07479a1cb08219ab719147e742e76090c9c773321959bb94946fffdd397a6437", + "sha256:07949c449a1abcf60d1ee6e88956d89404c7df3c8258f46589e912988e551987", + "sha256:105110f43a471dbd0060b9c9516cb8a6a79233631a04cc2ba16f28323ac6e025", + "sha256:11b74db56cdbe3cdee6e3f6982ecb70334fa10dce99ed58bf7894aaaa3b2a037", + "sha256:12b9c6996425c76ea6c457ace4f3073e715b8c545add07cd1a8f3a4f90691269", + "sha256:1489e263a8048bb8b6a8bac662eb2d402ea5d2b7b4699b72f385f1e2772db105", + "sha256:19736989797678c6af1e55cd49055cdbcb55d8f6b5583ac5335f933aba9101dc", + "sha256:1b4a266766514614f8aa60416e71f2fc6e575d36e7bdc90f644fadb2f4b75b95", + "sha256:2a8183b489dc1f7f80f135780fadc1108f14b31b8a40411c7a5b17425f65f28b", + "sha256:37fdb0d0111f1e2ff07139dfb79f1b49531f8e213c46f1163dd7642979b58c47", + "sha256:3f5735ffe4996d28b809371756219f5354864902a3b9e7c0b9ee87041209fc9c", + "sha256:49e7d93abdbd2990caced757e5fade25302f719c3c8fb6e6fff2dde98999fc41", + "sha256:5e34edd123674534acd70147f0ca331eaa2c74e6325fb2028c886aa26ba0b68c", + "sha256:62598a8a57f815db4c6259a4e97d857dab56697e7de8e8ab02352ab74da1995d", + "sha256:65c2c3add92b45fd0709db8594536aea39c2a67af0e27ffcf049c498501140b7", + "sha256:6ba6a53445bd3cfa809ef3ef5f1589aa6ba08784a1d962bf47d0940e871dab1c", + "sha256:6e7d61120573a7f2cd94cc095f9e81f6967c61ccdf194285aa143ecec8e0b708", + "sha256:7cec5b856506da6defb290f30c9ee687d5f5e8cb0bd3f6459dde43b0b4fa40ef", + "sha256:80b63928fa35083b33966ce1efb70e5b9607181e49dcd1c22c8c005e319f667f", + "sha256:82148ec5bddac30b51a5b3c1945075f896fa022cb93f8e4a01e9f6ee95292c5f", + "sha256:828743d939e9629bc267b8e2d08d8bb67cd4319c771a33d4b18b22dd8fb7440a", + "sha256:8d89f3976b10b4ce31118de72329025f70d2c6ead14a8217c5514dd2c6d5a78f", + "sha256:8eb5e1172eb569ea8a872796576e6a67c276351728b6455d5beb01242b027c6a", + "sha256:900131fafd8aead39ac7dd3a7e833be754c17a95cfd91221636949fe4eb0aa8a", + "sha256:910d11e1a385c654bf738bf3e6b8e6ed5de0f5610fcae2be9e5b398d8081d20e", + "sha256:910e1d2668e7de9648f2bcee30e180db2a6b15c30f887d7c4c93ddf96e3992e3", + "sha256:9aa87839c383bdbab6ef865787a1fb877af8dd03464c4400322726feaaadfc6d", + "sha256:a1b30560f2acc95aa8b2e06e716a13dbfc97314747b80d9707e307f77b40d6b3", + "sha256:a91296cb61e8df6f86d0c19cc4068228da256bf59bf86049fbd821084565327f", + "sha256:b42a28c1844fd9de8f3f7d540e36b66f3a9c83fceac7170ebc7a6a19edd9dcae", + "sha256:bd1c592e4d5974f0d08d4888e432157adba757c66da0246918e43677fafa2d30", + "sha256:c87f62a3d3b9888ed0fdde100ec06aa61ca9cd44bad9057d1dff9a516b5f5bb9", + "sha256:c99c003e088647b8a5b7c145d6f78c335f6348332b62e142d411c4b63d1460b9", + "sha256:ccdc4a71a4dabae05de219404f9f4abc38e3b58422177ff93d0da05967dafa07", + "sha256:d24fead1d4d076e1bfb006dcec392074a3cd8d7b4fc8a595aa64073b2b7a96ba", + "sha256:d58c3db7cd6eed54e6c06744db55456b65ebd7492ddeae9c1e93cfca7aa857d3", + "sha256:d764dcf130c428ef66786f866dd750f53182bc608813489915e9fc106bb0c82f", + "sha256:df2a58a472f332225671c35b0a830208b86d004f82baa8530fa3782c85646533", + "sha256:e722f16708d854fe924790e051061f6704a472c3bac347b6fd88033ea8dd0dc5", + "sha256:ecfed7367f965a0328cfbdd70da860f15441f002f613185668c6e6ebf5a0ac11", + "sha256:eeac2acb5a20ed25e0ad6d1df9891a520b78b404266b6d11778f25d5d691a6c9", + "sha256:f59e38625469987d7ef6d495323c55e7db6c212eaf6112267e0d3b565a2e9c9f", + "sha256:f89831ef99dd7dd169ab06d63a831adb9e20a87aac6d380266bbda5823349169", + "sha256:fd9192b7b70c573d7f214eb1ae35e00d359f6f5e4b27c7e21e30de1fc6204645" ], "markers": "python_version >= '3.9' and python_full_version not in '3.9.0, 3.9.1'", - "version": "==48.0.0" + "version": "==50.0.0" }, "cssselect2": { "hashes": [ @@ -801,6 +778,18 @@ "markers": "python_version >= '3.9'", "version": "==6.0.0" }, + "django-storages": { + "extras": [ + "google", + "s3" + ], + "hashes": [ + "sha256:11b7b6200e1cb5ffcd9962bd3673a39c7d6a6109e8096f0e03d46fab3d3aabd9", + "sha256:7a25ce8f4214f69ac9c7ce87e2603887f7ae99326c316bc8d2d75375e09341c9" + ], + "markers": "python_version >= '3.7'", + "version": "==1.14.6" + }, "djangoql": { "hashes": [ "sha256:51b3085a805627ebb43cfd0aa861137cdf8f69cc3c9244699718fe04a6c8e26d" @@ -1064,6 +1053,93 @@ "markers": "python_version >= '3.9'", "version": "==1.8.0" }, + "google-api-core": { + "hashes": [ + "sha256:98a779fe72de956eb1c9c2f47ff4c4432a668ece1a002ec38bed07ec2698ae59", + "sha256:cdf9c67e7ca2402d86ccbfde5f2503fc83e3cc3f58cc78456ae96cad24a6d2de" + ], + "markers": "python_version >= '3.10'", + "version": "==2.34.0" + }, + "google-auth": { + "hashes": [ + "sha256:40e229fc901f0a305b553050e5fce562d509bee0435be053abfa91582b51b90c", + "sha256:8ec438808f813ad034535000261eed1067475d229d05bbf4216e78c3f2362e53" + ], + "markers": "python_version >= '3.10'", + "version": "==2.56.3" + }, + "google-cloud-core": { + "hashes": [ + "sha256:1e044b131f2ae097b92312fa195164b0aeb6dc6a88e00231e1210516314c420c", + "sha256:2682a8a4474a32f56292fb4bca7fa7e4fb0b4af958f6abfe4bca8d195747fd45" + ], + "markers": "python_version >= '3.10'", + "version": "==2.6.1" + }, + "google-cloud-storage": { + "hashes": [ + "sha256:98208de6c21e85cecd3eb44551894efff33d98365500e178867d4305854a770a", + "sha256:a80bf8cac2794808aa61c50c5f769ecbbe2d10331bacd0d69d30e59b14b346b2" + ], + "markers": "python_version >= '3.10'", + "version": "==3.13.1" + }, + "google-crc32c": { + "hashes": [ + "sha256:014a7e68d623e9a4222d663931febc3033c5c7c9730785727de2a81f87d5bab8", + "sha256:01f126a5cfddc378290de52095e2c7052be2ba7656a9f0caf4bcd1bfb1833f8a", + "sha256:0470b8c3d73b5f4e3300165498e4cf25221c7eb37f1159e221d1825b6df8a7ff", + "sha256:119fcd90c57c89f30040b47c211acee231b25a45d225e3225294386f5d258288", + "sha256:14f87e04d613dfa218d6135e81b78272c3b904e2a7053b841481b38a7d901411", + "sha256:17446feb05abddc187e5441a45971b8394ea4c1b6efd88ab0af393fd9e0a156a", + "sha256:19b40d637a54cb71e0829179f6cb41835f0fbd9e8eb60552152a8b52c36cbe15", + "sha256:2a3dc3318507de089c5384cc74d54318401410f82aa65b2d9cdde9d297aca7cb", + "sha256:3b9776774b24ba76831609ffbabce8cdf6fa2bd5e9df37b594221c7e333a81fa", + "sha256:3cc0c8912038065eafa603b238abf252e204accab2a704c63b9e14837a854962", + "sha256:3d488e98b18809f5e322978d4506373599c0c13e6c5ad13e53bb44758e18d215", + "sha256:3ebb04528e83b2634857f43f9bb8ef5b2bbe7f10f140daeb01b58f972d04736b", + "sha256:450dc98429d3e33ed2926fc99ee81001928d63460f8538f21a5d6060912a8e27", + "sha256:4b8286b659c1335172e39563ab0a768b8015e88e08329fa5321f774275fc3113", + "sha256:57a50a9035b75643996fbf224d6661e386c7162d1dfdab9bc4ca790947d1007f", + "sha256:61f58b28e0b21fcb249a8247ad0db2e64114e201e2e9b4200af020f3b6242c9f", + "sha256:6f35aaffc8ccd81ba3162443fabb920e65b1f20ab1952a31b13173a67811467d", + "sha256:71734788a88f551fbd6a97be9668a0020698e07b2bf5b3aa26a36c10cdfb27b2", + "sha256:864abafe7d6e2c4c66395c1eb0fe12dc891879769b52a3d56499612ca93b6092", + "sha256:86cfc00fe45a0ac7359e5214a1704e51a99e757d0272554874f419f79838c5f7", + "sha256:87b0072c4ecc9505cfa16ee734b00cd7721d20a0f595be4d40d3d21b41f65ae2", + "sha256:87fa445064e7db928226b2e6f0d5304ab4cd0339e664a4e9a25029f384d9bb93", + "sha256:89c17d53d75562edfff86679244830599ee0a48efc216200691de8b02ab6b2b8", + "sha256:8b3f68782f3cbd1bce027e48768293072813469af6a61a86f6bb4977a4380f21", + "sha256:a428e25fb7691024de47fecfbff7ff957214da51eddded0da0ae0e0f03a2cf79", + "sha256:b0d1a7afc6e8e4635564ba8aa5c0548e3173e41b6384d7711a9123165f582de2", + "sha256:ba6aba18daf4d36ad4412feede6221414692f44d17e5428bdd81ad3fc1eee5dc", + "sha256:cb5c869c2923d56cb0c8e6bcdd73c009c36ae39b652dbe46a05eb4ef0ad01454", + "sha256:d511b3153e7011a27ab6ee6bb3a5404a55b994dc1a7322c0b87b29606d9790e2", + "sha256:db3fe8eaf0612fc8b20fa21a5f25bd785bc3cd5be69f8f3412b0ac2ffd49e733", + "sha256:e6584b12cb06796d285d09e33f63309a09368b9d806a551d8036a4207ea43697", + "sha256:f4b51844ef67d6cf2e9425983274da75f18b1597bb2c998e1c0a0e8d46f8f651", + "sha256:f639065ea2042d5c034bf258a9f085eaa7af0cd250667c0635a3118e8f92c69c" + ], + "markers": "python_version >= '3.9'", + "version": "==1.8.0" + }, + "google-resumable-media": { + "hashes": [ + "sha256:224975032ddb73f7ed9e2f0f4cc08ed1b06874c52d48cc8533e3eb72980b21a0", + "sha256:4e2cbc704207ddc09f23b1f18e8ef4a4ccbfe0f1768b370e5c969704adbd0a1c" + ], + "markers": "python_version >= '3.10'", + "version": "==2.10.1" + }, + "googleapis-common-protos": { + "hashes": [ + "sha256:28a1934bcd33b9c9da66ac301a0a4227e3367f095a17d0375cb98f0a09d93b79", + "sha256:d3042c6c5a2d4e67113104d6b6818b59b6bd92a197f2a91508e801fe815cf071" + ], + "markers": "python_version >= '3.10'", + "version": "==1.75.1" + }, "gunicorn": { "hashes": [ "sha256:ec400d38950de4dfd418cff8328b2c8faed0edb0d517d3394e457c317908ca4d", @@ -1200,11 +1276,11 @@ }, "idna": { "hashes": [ - "sha256:048adeaf8c2d788c40fee287673ccaa74c24ffd8dcf09ffa555a2fbb59f10ac8", - "sha256:ca962446ea538f7092a95e057da437618e886f4d349216d2b1e294abfdb65fdc" + "sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2", + "sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848" ], - "markers": "python_version >= '3.8'", - "version": "==3.15" + "markers": "python_version >= '3.9'", + "version": "==3.18" }, "inflection": { "hashes": [ @@ -1777,6 +1853,28 @@ "markers": "python_version >= '3.10'", "version": "==0.5.2" }, + "proto-plus": { + "hashes": [ + "sha256:5f91b30dafa6bb38d432c5557a6ee1d35ffd40b4b1e0e3ca27260448560b91d9", + "sha256:dc76880b8ee951cca002098574376cf71e055f9f16d9ba6570fb8a06f726d281" + ], + "markers": "python_version >= '3.10'", + "version": "==1.28.3" + }, + "protobuf": { + "hashes": [ + "sha256:11d6b0ec246892d85215b0a13ca6e0233cf5284b68f0ac02646427f4ff88a799", + "sha256:230a75ddfc2de4806e56696ce9640c1cdfdb6543b7cfce98d42a4c0a0e7bdb87", + "sha256:24f857477359a85c0c235261b8ba905fd51b2562f4a64ca1df5473f29850cbf6", + "sha256:353652e4efd0bca5b5fc2656abf8307ef351f0cf938c9eba09f0e09c20a25c30", + "sha256:4bc97768d8fe4ad6743c8a19403e314511ed9f6d13205b687e52421c023ac1b9", + "sha256:74758715c53d7158fb76caf4f0cfdacc5329a4b1bb994f865d6cf302d413a1c4", + "sha256:b73f9489a4b8b1c9cb1f8ed951c736392592edb24b9d6819f36d2e10b171d5b4", + "sha256:ce115a26fe0c39a2c29973d914d327e516a6455464489fe3cd1e51a1b354f81a" + ], + "markers": "python_version >= '3.10'", + "version": "==7.35.1" + }, "psycopg": { "extras": [ "c" @@ -1802,6 +1900,22 @@ ], "version": "==1.9.4" }, + "pyasn1": { + "hashes": [ + "sha256:9c447d8431c947fe4c8febc4ed9e760bc29011a5b01e5c74b67025bd9fb8ce81", + "sha256:deda9277cfd454080ec40b207fb6df82206a3a2688735233cdcd8d3d565f088b" + ], + "markers": "python_version >= '3.8'", + "version": "==0.6.4" + }, + "pyasn1-modules": { + "hashes": [ + "sha256:29253a9207ce32b64c3ac6600edc75368f98473906e8fd1043bd6b5b1de2c14a", + "sha256:677091de870a80aae844b1ca6134f54652fa2c8c5a52aa396440ac3106e941e6" + ], + "markers": "python_version >= '3.8'", + "version": "==0.4.2" + }, "pycparser": { "hashes": [ "sha256:600f49d217304a5902ac3c37e1281c9fe94e4d0489de643a9504c5cdfdfc6b29", @@ -2267,11 +2381,11 @@ }, "s3transfer": { "hashes": [ - "sha256:9edeb6d1c3c2f89d6050348548834ad8289610d886e5bf7b7207728bd43ce33a", - "sha256:ce3801712acf4ad3e89fb9990df97b4972e93f4b3b0004d214be5bce12814c20" + "sha256:042dd5e3b1b512355e35a23f0223e426b7042e80b97830ea2680ddce327fc45e", + "sha256:5b9827d1044159bbb01b86ef8902760ea39281927f5de31de75e1d657177bf4c" ], "markers": "python_version >= '3.10'", - "version": "==0.17.0" + "version": "==0.17.1" }, "sentry-sdk": { "hashes": [ diff --git a/care/emr/api/viewsets/file_assets.py b/care/emr/api/viewsets/file_assets.py new file mode 100644 index 0000000000..25a1567445 --- /dev/null +++ b/care/emr/api/viewsets/file_assets.py @@ -0,0 +1,62 @@ +""" +Public asset delivery for facility cover images and user avatars. + +ADR-0001: a client never receives a storage-provider URL. These two objects were +already world-readable directly from the bucket, so the routes stay +unauthenticated — who can see the image is unchanged. What changes is that CARE +serves the bytes, which lets the bucket become private and keeps the provider +interchangeable. + +They are separate views rather than actions on FacilityViewSet / UserViewSet +because those viewsets filter their querysets by ``request.user`` and cannot +serve an anonymous request. +""" + +from django.core.files.storage import storages +from rest_framework.exceptions import NotFound +from rest_framework.views import APIView + +from care.emr.utils.file_download import storage_file_response +from care.facility.models.facility import Facility +from care.users.models import User +from care.utils.file_uploads.cover_image import STORAGE_ALIAS +from care.utils.shortcuts import get_object_or_404 + +#: ``upload_cover_image`` mints a fresh key containing a random token for every +#: upload, so the bytes behind a given key never change -- replacing an image +#: produces a different key. The response is therefore immutable per key and +#: safe for anonymous browser, CDN and reverse-proxy caches, which is what CARE +#: serving the bytes would otherwise cost us over reading the bucket directly. +PUBLIC_ASSET_CACHE_CONTROL = "public, max-age=31536000, immutable" + + +class PublicAssetView(APIView): + """Unauthenticated read-only delivery of a public image.""" + + authentication_classes = () + permission_classes = () + + @staticmethod + def serve(object_key: str | None): + if not object_key: + msg = "No image set" + raise NotFound(msg) + response = storage_file_response( + storages[STORAGE_ALIAS], + object_key, + filename=object_key.rsplit("/", 1)[-1], + ) + response.headers["Cache-Control"] = PUBLIC_ASSET_CACHE_CONTROL + return response + + +class FacilityCoverImageView(PublicAssetView): + def get(self, request, external_id): + facility = get_object_or_404(Facility, external_id=external_id) + return self.serve(facility.cover_image_url) + + +class UserProfilePictureView(PublicAssetView): + def get(self, request, username): + user = get_object_or_404(User, username=username, deleted=False) + return self.serve(user.profile_picture_url) diff --git a/care/emr/api/viewsets/file_upload.py b/care/emr/api/viewsets/file_upload.py index f91ad5041c..81199fca40 100644 --- a/care/emr/api/viewsets/file_upload.py +++ b/care/emr/api/viewsets/file_upload.py @@ -1,16 +1,16 @@ -import base64 - import magic from django.conf import settings -from django.core.files.base import ContentFile from django.db import transaction from django.utils import timezone from django_filters import rest_framework as filters -from drf_spectacular.utils import extend_schema +from drf_spectacular.types import OpenApiTypes +from drf_spectacular.utils import extend_schema, extend_schema_field from pydantic import BaseModel from rest_framework import filters as rest_framework_filters +from rest_framework import serializers from rest_framework.decorators import action -from rest_framework.exceptions import PermissionDenied, ValidationError +from rest_framework.exceptions import NotFound, PermissionDenied, ValidationError +from rest_framework.parsers import MultiPartParser from rest_framework.response import Response from care.emr.api.viewsets.base import ( @@ -25,12 +25,14 @@ from care.emr.models.diagnostic_report import DiagnosticReport from care.emr.models.service_request import ServiceRequest from care.emr.resources.file_upload.spec import ( + FileCategoryChoices, FileTypeChoices, FileUploadCreateSpec, FileUploadListSpec, FileUploadRetrieveSpec, FileUploadUpdateSpec, ) +from care.emr.utils.file_download import file_object_response from care.security.authorization import AuthorizationController from care.utils.shortcuts import get_object_or_404 @@ -116,6 +118,55 @@ class FileUploadFilter(filters.FilterSet): name = filters.CharFilter(field_name="name", lookup_expr="icontains") +@extend_schema_field(OpenApiTypes.BINARY) +class BinaryFileField(serializers.FileField): + """ + A `FileField` that documents itself as binary. + + drf-spectacular renders `FileField` as `format: uri` by default, because + DRF serialises it to a URL on *output*. In a multipart request body it is + raw bytes, so the annotation is applied here rather than by flipping + COMPONENT_SPLIT_REQUEST, which would reshape every schema in the project. + """ + + +class FileUploadMultipartSerializer(serializers.Serializer): + """ + `multipart/form-data` upload request (ADR-0002). + + Declares the transport contract and the binary field. The metadata fields + are carried through to `FileUploadCreateSpec`, which remains the + authoritative validator for the logical file type, category and filename. + + `file_type` and `file_category` are `ChoiceField`s here as well, so the + choices appear in the generated schema and a bad value is rejected before + the file is read. That duplicates the spec's enums deliberately: both are + generated from the same `FileTypeChoices` / `FileCategoryChoices`, so adding + a member updates both. No rule is restated by hand. + """ + + file = BinaryFileField( + help_text="The file itself, sent as a normal multipart file part." + ) + name = serializers.CharField(help_text="Display name for the file.") + associating_id = serializers.CharField( + help_text="External id of the object the file belongs to." + ) + file_type = serializers.ChoiceField( + choices=[choice.value for choice in FileTypeChoices] + ) + file_category = serializers.ChoiceField( + choices=[choice.value for choice in FileCategoryChoices] + ) + original_name = serializers.CharField( + required=False, + help_text=( + "Original filename. Defaults to the uploaded part's filename; " + "supply it only to override." + ), + ) + + class FileUploadViewSet( EMRCreateMixin, EMRRetrieveMixin, EMRUpdateMixin, EMRListMixin, EMRBaseViewSet ): @@ -173,6 +224,21 @@ def get_queryset(self): file_authorizer(self.request.user, obj.file_type, obj.associating_id, "read") return super().get_queryset() + @extend_schema( + description="Download the file through CARE. Reads through Django Storage; " + "no storage-provider URL is exposed.", + responses={(200, "application/octet-stream"): OpenApiTypes.BINARY}, + ) + @action(detail=True, methods=["GET"]) + def download(self, request, *args, **kwargs): + # get_object() -> get_queryset(), which runs file_authorizer(..., "read") + # for every detail action. + obj = self.get_object() + if not obj.upload_completed: + msg = "File upload is not complete" + raise NotFound(msg) + return file_object_response(obj) + @extend_schema(responses={200: FileUploadListSpec}) @action(detail=True, methods=["POST"]) def mark_upload_completed(self, request, *args, **kwargs): @@ -210,31 +276,44 @@ def archive(self, request, *args, **kwargs): ) return Response(FileUploadListSpec.serialize(obj).to_json()) - @action(detail=False, methods=["POST"], url_path="upload-file") + @extend_schema( + description=( + "Upload a file through CARE using multipart/form-data. The bytes " + "are streamed to the configured storage backend through Django " + "Storage; no storage-provider URL is involved." + ), + request={"multipart/form-data": FileUploadMultipartSerializer}, + responses={200: FileUploadRetrieveSpec}, + ) + @action( + detail=False, + methods=["POST"], + url_path="upload-file", + parser_classes=[MultiPartParser], + ) def upload_file(self, request, *args, **kwargs): - file_name = request.data.get("original_name") - file_data = request.data.get("file_data") + serializer = FileUploadMultipartSerializer(data=request.data) + serializer.is_valid(raise_exception=True) + payload = serializer.validated_data - if not file_name or not file_data: - raise ValidationError( - "Missing required fields: 'original_name' or 'file_data'" - ) - - try: - file_content = base64.b64decode(file_data) - except Exception as e: - error = "Invalid base64-encoded file data" - raise ValidationError(error) from e - - uploaded_file = ContentFile(file_content, name=file_name) + # Django's upload handlers have already decided whether this is an + # InMemoryUploadedFile or a TemporaryUploadedFile, based on + # FILE_UPLOAD_MAX_MEMORY_SIZE. Either way it is a file-like object and + # is never fully materialised here. + uploaded_file = payload["file"] + file_name = payload.get("original_name") or uploaded_file.name max_file_size = settings.MAX_FILE_UPLOAD_SIZE * 1024 * 1024 if uploaded_file.size > max_file_size: error = f"File size exceeds the limit of {max_file_size / (1024 * 1024)}MB" raise ValidationError(error) + # Sniff the declared type from the leading bytes rather than trusting + # the part's Content-Type header, matching the previous behaviour. try: - mime_type = magic.from_buffer(file_content[:2048], mime=True) + header = uploaded_file.read(2048) + uploaded_file.seek(0) + mime_type = magic.from_buffer(header, mime=True) except Exception as e: error = "Error detecting file type." raise ValidationError(error) from e @@ -245,26 +324,38 @@ def upload_file(self, request, *args, **kwargs): request_data = { "original_name": file_name, - "name": request.data.get("name"), - "associating_id": request.data.get("associating_id"), - "file_type": request.data.get("file_type"), - "file_category": request.data.get("file_category"), + "name": payload["name"], + "associating_id": payload["associating_id"], + "file_type": payload["file_type"], + "file_category": payload["file_category"], "mime_type": mime_type, } + # The row is written first and the object second, both inside one + # transaction: a storage failure rolls the row back, so no completed + # record can exist without its object. The reverse gap — object written, + # commit fails — leaves an orphan object, which is pre-existing + # behaviour recorded as B8 in unresolved-items.md. with transaction.atomic(): file_upload = FileUploadCreateSpec(**request_data).de_serialize() file_upload._just_created = False # noqa SLF001 self.authorize_create(file_upload) file_upload.save() + # Only the storage write is translated into "failed to upload to + # storage". The save below is deliberately outside the block: a + # database failure there is not a storage failure, and reporting it + # as one sends whoever reads the log to the wrong system. try: + # The UploadedFile is handed straight to Django Storage; it is + # not read into memory first. file_upload.files_manager.put_object(file_upload, uploaded_file) - file_upload.upload_completed = True - file_upload.updated_by = request.user - file_upload.save(skip_internal_name=True) except Exception as e: error_msg = "Failed to upload file to storage" raise ValidationError(error_msg) from e + file_upload.upload_completed = True + file_upload.updated_by = request.user + file_upload.save(skip_internal_name=True) + return Response(FileUploadRetrieveSpec.serialize(file_upload).to_json()) diff --git a/care/emr/api/viewsets/report/report_upload.py b/care/emr/api/viewsets/report/report_upload.py index 4c0e3cb482..c47cc5c6b2 100644 --- a/care/emr/api/viewsets/report/report_upload.py +++ b/care/emr/api/viewsets/report/report_upload.py @@ -3,6 +3,7 @@ from django.utils import timezone from django_filters import BooleanFilter, CharFilter, FilterSet from django_filters.rest_framework import DjangoFilterBackend +from drf_spectacular.types import OpenApiTypes from drf_spectacular.utils import extend_schema from pydantic import UUID4, BaseModel, field_validator from rest_framework import status @@ -26,6 +27,7 @@ ReportUploadRetrieveSpec, ) from care.emr.tasks.report_generation import generate_report_task +from care.emr.utils.file_download import file_object_response from care.security.authorization.base import AuthorizationController from care.utils.shortcuts import get_object_or_404 @@ -94,6 +96,20 @@ def authorize_update(self, request_obj, model_instance): self.request.user, model_instance.report_type, model_instance.associating_id ) + @extend_schema( + description="Download the report through CARE. Reads through Django Storage; " + "no storage-provider URL is exposed.", + responses={(200, "application/octet-stream"): OpenApiTypes.BINARY}, + tags=["report"], + ) + @action(detail=True, methods=["GET"]) + def download(self, request, *args, **kwargs): + obj = self.get_object() + # get_queryset() only authorizes the list action, so a detail action + # must authorize explicitly or it would serve any report to any user. + read_report_authorizer(request.user, obj.report_type, obj.associating_id) + return file_object_response(obj) + @extend_schema( description="Generate a report from a template with patient/encounter data", request=GenerateReportRequest, diff --git a/care/emr/models/file_upload.py b/care/emr/models/file_upload.py index e71837fcbf..ebfdd54a3d 100644 --- a/care/emr/models/file_upload.py +++ b/care/emr/models/file_upload.py @@ -4,9 +4,8 @@ from django.db import models from care.emr.models import EMRBaseModel -from care.emr.utils.file_manager import S3FilesManager +from care.emr.utils.file_manager import FilesManager from care.users.models import User -from care.utils.csp.config import BucketType from care.utils.models.validators import parse_file_extension @@ -30,7 +29,7 @@ class FileUpload(EMRBaseModel): related_name="archived_files", ) - files_manager = S3FilesManager(BucketType.PATIENT) + files_manager = FilesManager("patient") def get_extension(self): extensions = parse_file_extension(self.internal_name) diff --git a/care/emr/models/report/report_upload.py b/care/emr/models/report/report_upload.py index 3c10182344..36d56def14 100644 --- a/care/emr/models/report/report_upload.py +++ b/care/emr/models/report/report_upload.py @@ -4,9 +4,8 @@ from django.db import models from care.emr.models import EMRBaseModel -from care.emr.utils.file_manager import S3FilesManager +from care.emr.utils.file_manager import FilesManager from care.users.models import User -from care.utils.csp.config import BucketType from care.utils.models.validators import parse_file_extension @@ -31,11 +30,11 @@ class ReportUpload(EMRBaseModel): related_name="archived_reports", ) - files_manager = S3FilesManager(BucketType.REPORT) + files_manager = FilesManager("report") @property def file_type(self): - """Alias for report_type to maintain compatibility with S3FilesManager""" + """Alias for report_type, so reports share the storage-name convention""" return self.report_type def get_extension(self): diff --git a/care/emr/reports/context_builder/data_points/fileupload.py b/care/emr/reports/context_builder/data_points/fileupload.py index 27a9310517..30178e23a7 100644 --- a/care/emr/reports/context_builder/data_points/fileupload.py +++ b/care/emr/reports/context_builder/data_points/fileupload.py @@ -6,6 +6,7 @@ Field, QuerysetContextBuilder, ) +from care.emr.utils.file_download import file_download_url class FileUploadReportFilter(filters.FilterSet): @@ -26,9 +27,7 @@ class FileUploadContextBuilder(QuerysetContextBuilder): url = Field( display="File URL", preview_value="https://s3.amazonaws.com/bucket/patient/12345/file.pdf", - mapping=lambda f: f.files_manager.read_signed_url(f) - if f.upload_completed - else None, + mapping=lambda f: file_download_url(f) if f.upload_completed else None, description="URL to access the uploaded file", ) diff --git a/care/emr/reports/report_utils.py b/care/emr/reports/report_utils.py index ae36dc49ab..6f15db7381 100644 --- a/care/emr/reports/report_utils.py +++ b/care/emr/reports/report_utils.py @@ -3,6 +3,7 @@ from uuid import uuid4 from django.core.cache import cache +from django.core.files.base import ContentFile from django.utils import timezone from care.emr.models.report.report_upload import ReportUpload @@ -121,8 +122,10 @@ def generate_and_upload_report( # noqa:PLR0915 report_upload.save(skip_internal_name=True) try: + # output_bytes is already fully materialised by the renderer above; see + # docs/xii/architecture/inventory/storage-call-sites.md section 5. report_upload.files_manager.put_object( - report_upload, output_bytes, ContentType=mime_type + report_upload, ContentFile(output_bytes), content_type=mime_type ) report_upload.upload_completed = True report_upload.save() diff --git a/care/emr/resources/file_upload/spec.py b/care/emr/resources/file_upload/spec.py index d0b4647b56..aa40b74ec7 100644 --- a/care/emr/resources/file_upload/spec.py +++ b/care/emr/resources/file_upload/spec.py @@ -8,6 +8,7 @@ from care.emr.models import FileUpload from care.emr.resources.base import EMRResource, model_from_cache from care.emr.resources.user.spec import UserSpec +from care.emr.utils.file_download import file_download_url from care.utils.models.validators import file_name_validator @@ -103,18 +104,20 @@ def perform_extra_serialization(cls, mapping, obj): class FileUploadRetrieveSpec(FileUploadListSpec): - signed_url: str | None = None - read_signed_url: str | None = None + # ADR-0001: CARE mediates all object transport. This is a CARE route, never + # a storage-provider URL. Uploads go to POST /api/v1/files/upload-file/. + download_url: str | None = None internal_name: str # Not sure if this needs to be returned @classmethod def perform_extra_serialization(cls, mapping, obj): super().perform_extra_serialization(mapping, obj) - if getattr(obj, "_just_created", False): - # Calculate Write URL and return it - mapping["signed_url"] = obj.files_manager.signed_url(obj) - else: - mapping["read_signed_url"] = obj.files_manager.read_signed_url(obj) + # A row that has not completed its upload has no object in storage yet, + # so a download route would only ever 404. Advertise it once the bytes + # are there; the download action refuses incomplete rows to match. + mapping["download_url"] = ( + file_download_url(obj) if obj.upload_completed else None + ) class ConsentFileUploadCreateSpec(FileUploadBaseSpec): diff --git a/care/emr/resources/report/report_upload/spec.py b/care/emr/resources/report/report_upload/spec.py index e61f546537..e3a593e0c8 100644 --- a/care/emr/resources/report/report_upload/spec.py +++ b/care/emr/resources/report/report_upload/spec.py @@ -7,6 +7,7 @@ from care.emr.resources.base import EMRResource from care.emr.resources.report.template.spec import TemplateReadSpec from care.emr.resources.user.spec import UserSpec +from care.emr.utils.file_download import report_download_url class ReportUploadBaseSpec(EMRResource): @@ -41,14 +42,11 @@ def perform_extra_serialization(cls, mapping, obj): class ReportUploadRetrieveSpec(ReportUploadListSpec): - signed_url: str | None = None - read_signed_url: str | None = None + # ADR-0001: a CARE route, never a storage-provider URL. + download_url: str | None = None internal_name: str @classmethod def perform_extra_serialization(cls, mapping, obj): super().perform_extra_serialization(mapping, obj) - if getattr(obj, "_just_created", False): - mapping["signed_url"] = obj.files_manager.signed_url(obj) - else: - mapping["read_signed_url"] = obj.files_manager.read_signed_url(obj) + mapping["download_url"] = report_download_url(obj) diff --git a/care/emr/tasks/cleanup_incomplete_file_uploads.py b/care/emr/tasks/cleanup_incomplete_file_uploads.py index 5f509302d3..efa3eff33f 100644 --- a/care/emr/tasks/cleanup_incomplete_file_uploads.py +++ b/care/emr/tasks/cleanup_incomplete_file_uploads.py @@ -30,7 +30,7 @@ def cleanup_incomplete_file_uploads(): for file in queryset: if file.internal_name: try: - file_manager.delete_object(file, quiet=True) + file_manager.delete_object(file) except Exception as e: logger.error( "Failed to delete file upload object %s: %s", diff --git a/care/emr/tests/test_file_upload_api.py b/care/emr/tests/test_file_upload_api.py index ab6cd26aa6..b306e4e4ae 100644 --- a/care/emr/tests/test_file_upload_api.py +++ b/care/emr/tests/test_file_upload_api.py @@ -1,11 +1,9 @@ -import base64 import io from datetime import timedelta -import requests -from botocore.exceptions import ClientError from django.conf import settings -from django.test import override_settings +from django.core.files.base import ContentFile +from django.core.files.uploadedfile import SimpleUploadedFile from django.urls import reverse from PIL import Image @@ -13,10 +11,9 @@ from care.emr.tasks.cleanup_incomplete_file_uploads import ( cleanup_incomplete_file_uploads, ) -from care.utils.tests.base import CareAPITestBase +from care.utils.tests.base import CareAPITestBase, response_content -@override_settings(FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT=settings.BUCKET_ENDPOINT) class FileUploadTestCase(CareAPITestBase): def setUp(self): super().setUp() @@ -34,6 +31,20 @@ def setUp(self): self.client.force_authenticate(user=self.user) + def upload_payload(self, **overrides): + """A valid multipart upload body (ADR-0002).""" + payload = { + "file": SimpleUploadedFile( + self.file.name, self.file.getvalue(), content_type=self.file_mime_type + ), + "name": "file", + "file_type": "patient", + "file_category": "unspecified", + "associating_id": str(self.patient.external_id), + } + payload.update(overrides) + return payload + def test_upload_user_avatar(self): url = reverse("users-profile-picture", args=[self.user.username]) response = self.client.post( @@ -57,88 +68,45 @@ def test_upload_facility_cover_image(self): self.assertTrue(self.facility.cover_image_url) def test_upload_patient_file(self): - url = reverse("files-list") - - response = self.client.post( - url, - { - "name": "file", - "original_name": "file.jpg", - "file_type": "patient", - "file_category": "unspecified", - "associating_id": str(self.patient.external_id), - "mime_type": self.file_mime_type, - }, - format="json", - ) - self.assertEqual(response.status_code, 200, response.data) - - file_upload_response = requests.put( - response.data["signed_url"], - data=self.file, - headers={ - "Content-Type": self.file_mime_type, - "x-ms-blob-type": "BlockBlob", - }, - timeout=5, - ) - self.assertIn( - file_upload_response.status_code, [200, 201], file_upload_response.text - ) - - response = self.client.post( - reverse("files-mark-upload-completed", args=[response.data["id"]]), - format="json", + upload = self.client.post( + reverse("files-upload-file"), + self.upload_payload(), + format="multipart", ) - self.assertEqual(response.status_code, 200, response.data) + self.assertEqual(upload.status_code, 200, upload.data) - response = self.client.get( - reverse("files-detail", args=[response.data["id"]]), - format="json", - ) - self.assertEqual(response.status_code, 200, response.data) + detail = self.client.get(reverse("files-detail", args=[upload.data["id"]])) + self.assertEqual(detail.status_code, 200, detail.data) - file_response = requests.get( - response.data["read_signed_url"], - timeout=5, - ) - self.assertEqual(file_response.status_code, 200, file_response.text) - self.assertEqual(file_response.content, self.file.getvalue()) + # The download URL is a CARE route, not a storage-provider URL. + download_url = detail.data["download_url"] self.assertEqual( - file_response.headers["Content-Type"], - self.file_mime_type, - file_response.headers, + download_url, reverse("files-download", args=[upload.data["id"]]) ) - # NOTE: azure does not support content-disposition + + file_response = self.client.get(download_url) + self.assertEqual(file_response.status_code, 200) + self.assertEqual(response_content(file_response), self.file.getvalue()) + self.assertEqual(file_response.headers["Content-Type"], self.file_mime_type) + # An inline-safe type still renders inline, as the presigned + # ResponseContentDisposition used to arrange. self.assertEqual( file_response.headers["Content-Disposition"], - f"inline; filename={self.file.name}", + f'inline; filename="{self.file.name}"', file_response.headers, ) def test_direct_file_upload(self): - url = reverse("files-upload-file") response = self.client.post( - url, - { - "name": "file", - "original_name": "file.jpg", - "file_type": "patient", - "file_category": "unspecified", - "associating_id": str(self.patient.external_id), - "mime_type": self.file_mime_type, - "file_data": base64.b64encode(self.file.read()).decode("utf-8"), - }, + reverse("files-upload-file"), + self.upload_payload(), format="multipart", ) self.assertEqual(response.status_code, 200, response.data) - file_response = requests.get( - response.data["read_signed_url"], - timeout=5, - ) - self.assertEqual(file_response.status_code, 200, file_response.text) - self.assertEqual(file_response.content, self.file.getvalue()) + file_response = self.client.get(response.data["download_url"]) + self.assertEqual(file_response.status_code, 200) + self.assertEqual(response_content(file_response), self.file.getvalue()) def test_cleanup_incomplete_file_uploads(self): url = reverse("files-list") @@ -161,24 +129,17 @@ def test_cleanup_incomplete_file_uploads(self): ) file_obj.save() - file_upload_response = requests.put( - response.data["signed_url"], - data=self.file, - headers={ - "Content-Type": self.file_mime_type, - "x-ms-blob-type": "BlockBlob", - }, - timeout=5, - ) - self.assertIn( - file_upload_response.status_code, [200, 201], file_upload_response.text - ) + # Put a real object behind the incomplete row, through Django Storage. + file_obj.files_manager.put_object(file_obj, ContentFile(self.file.getvalue())) + self.assertTrue(file_obj.files_manager.exists(file_obj)) cleanup_incomplete_file_uploads.delay() - with self.assertRaises(ClientError) as ce: + # Provider-neutral assertions: the object is gone, and opening it + # raises the standard Django Storage error rather than a boto3 one. + self.assertFalse(file_obj.files_manager.exists(file_obj)) + with self.assertRaises(FileNotFoundError): file_obj.files_manager.get_object(file_obj) - self.assertEqual(ce.exception.response["Error"]["Code"], "NoSuchKey") with self.assertRaises(FileUpload.DoesNotExist): file_obj.refresh_from_db() diff --git a/care/emr/tests/test_file_upload_multipart.py b/care/emr/tests/test_file_upload_multipart.py new file mode 100644 index 0000000000..2af2c01e6f --- /dev/null +++ b/care/emr/tests/test_file_upload_multipart.py @@ -0,0 +1,447 @@ +""" +Multipart upload transport tests (ADR-0002 / ES-02). + +Covers the HTTP transport contract: what the endpoint accepts, what it rejects, +how it behaves under failure, and that it stays provider-neutral. Persistence +behaviour itself is covered by the ES-01 storage tests. +""" + +import io +from unittest.mock import patch + +from django.core.files.uploadedfile import ( + InMemoryUploadedFile, + SimpleUploadedFile, + TemporaryUploadedFile, +) +from django.test import override_settings +from django.urls import reverse +from PIL import Image + +from care.emr.models.file_upload import FileUpload +from care.emr.utils.file_manager import FilesManager, get_storage_name +from care.utils.tests.base import CareAPITestBase, response_content + + +def jpeg_bytes(size=(800, 800)) -> bytes: + buffer = io.BytesIO() + Image.new("RGB", size).save(buffer, format="JPEG") + return buffer.getvalue() + + +class MultipartUploadTestBase(CareAPITestBase): + def setUp(self): + super().setUp() + self.user = self.create_super_user() + self.facility = self.create_facility(user=self.user) + self.patient = self.create_patient() + self.payload_bytes = jpeg_bytes() + self.url = reverse("files-upload-file") + self.client.force_authenticate(user=self.user) + + def upload( + self, + *, + content=None, + filename="scan.jpg", + content_type="image/jpeg", + **overrides, + ): + body = { + "file": SimpleUploadedFile( + filename, + self.payload_bytes if content is None else content, + content_type=content_type, + ), + "name": "scan", + "file_type": "patient", + "file_category": "unspecified", + "associating_id": str(self.patient.external_id), + } + body.update(overrides) + body = {k: v for k, v in body.items() if v is not None} + return self.client.post(self.url, body, format="multipart") + + +class SuccessfulUploadTests(MultipartUploadTestBase): + def test_upload_succeeds(self): + response = self.upload() + self.assertEqual(response.status_code, 200, response.data) + + def test_record_is_correct(self): + response = self.upload() + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + self.assertEqual(file_obj.name, "scan") + self.assertEqual(file_obj.file_type, "patient") + self.assertEqual(file_obj.file_category, "unspecified") + self.assertEqual(file_obj.associating_id, str(self.patient.external_id)) + self.assertEqual(file_obj.meta["mime_type"], "image/jpeg") + self.assertTrue(file_obj.upload_completed) + + def test_original_name_defaults_to_the_uploaded_filename(self): + response = self.upload(filename="referral.jpg") + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + # internal_name is regenerated but keeps the extension. + self.assertTrue(file_obj.internal_name.endswith(".jpg")) + + def test_explicit_original_name_is_honoured(self): + response = self.upload(filename="scan.jpg", original_name="override.jpeg") + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + self.assertTrue(file_obj.internal_name.endswith(".jpeg")) + + def test_uses_the_patient_alias_and_es01_object_name(self): + response = self.upload() + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + self.assertEqual(file_obj.files_manager.storage_alias, "patient") + # The ES-01 naming helper remains authoritative; transport adds none. + self.assertEqual( + get_storage_name(file_obj), f"patient/{file_obj.internal_name}" + ) + self.assertTrue(file_obj.files_manager.exists(file_obj)) + + def test_stored_bytes_match_and_download_works(self): + response = self.upload() + download = self.client.get(response.data["download_url"]) + self.assertEqual(download.status_code, 200) + self.assertEqual(response_content(download), self.payload_bytes) + + def test_response_is_provider_neutral(self): + response = self.upload() + body = str(response.data) + for forbidden in ( + "signed_url", + "read_signed_url", + "minio", + "amazonaws", + "storage.googleapis", + "X-Amz-Signature", + ":9100", + "patient-bucket", + ): + self.assertNotIn(forbidden, body, forbidden) + + def test_response_exposes_no_bucket_or_endpoint_key(self): + response = self.upload() + for key in ("bucket", "endpoint", "signed_url", "read_signed_url"): + self.assertNotIn(key, response.data) + + +class RejectionTests(MultipartUploadTestBase): + def test_missing_file_is_rejected(self): + response = self.upload(file=None) + self.assertEqual(response.status_code, 400, response.data) + + def test_missing_metadata_is_rejected(self): + response = self.upload(associating_id=None) + self.assertEqual(response.status_code, 400, response.data) + + def test_base64_payload_is_no_longer_accepted(self): + # The old contract must not work, even by accident. + response = self.client.post( + self.url, + { + "name": "scan", + "original_name": "scan.jpg", + "file_type": "patient", + "file_category": "unspecified", + "associating_id": str(self.patient.external_id), + "file_data": "/9j/4AAQSkZJRgABAQAAAQABAAD//gA7Q1JFQVRPUg==", + }, + format="multipart", + ) + self.assertEqual(response.status_code, 400, response.data) + + @override_settings(MAX_FILE_UPLOAD_SIZE=1) + def test_oversized_file_is_rejected(self): + response = self.upload(content=b"x" * (2 * 1024 * 1024)) + self.assertEqual(response.status_code, 400, response.data) + self.assertIn("size", str(response.data).lower()) + + def test_oversized_file_is_not_persisted(self): + with override_settings(MAX_FILE_UPLOAD_SIZE=1): + self.upload(content=b"x" * (2 * 1024 * 1024)) + self.assertFalse(FileUpload.objects.exists()) + + +class ValidationTests(MultipartUploadTestBase): + """MIME is sniffed from content; the declared part header is not trusted.""" + + def test_allowed_mime_accepted(self): + self.assertEqual(self.upload().status_code, 200) + + def test_disallowed_content_rejected_despite_allowed_declared_type(self): + # Declares image/jpeg but the bytes are a shell script. + response = self.upload( + content=b"#!/bin/sh\necho pwned\n", + filename="payload.jpg", + content_type="image/jpeg", + ) + self.assertEqual(response.status_code, 400, response.data) + self.assertIn("not allowed", str(response.data)) + + def test_declared_mime_is_ignored_in_favour_of_content(self): + # Declares an unsafe type but the bytes are a real JPEG. + response = self.upload(content=jpeg_bytes(), content_type="application/x-sh") + self.assertEqual(response.status_code, 200, response.data) + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + self.assertEqual(file_obj.meta["mime_type"], "image/jpeg") + + def test_missing_declared_mime_is_fine(self): + response = self.upload(content_type="") + self.assertEqual(response.status_code, 200, response.data) + + def test_blocked_extension_is_rejected(self): + response = self.upload(filename="payload.exe") + self.assertEqual(response.status_code, 400, response.data) + + def test_uppercase_extension_is_accepted(self): + response = self.upload(filename="SCAN.JPG") + self.assertEqual(response.status_code, 200, response.data) + + def test_double_extension_uses_the_outermost(self): + response = self.upload(filename="scan.jpg.exe") + self.assertEqual(response.status_code, 400, response.data) + + def test_unknown_file_type_is_rejected(self): + response = self.upload(file_type="not_a_type") + self.assertEqual(response.status_code, 400, response.data) + + def test_unknown_file_category_is_rejected(self): + response = self.upload(file_category="not_a_category") + self.assertEqual(response.status_code, 400, response.data) + + +class AuthorizationTests(MultipartUploadTestBase): + def test_unauthenticated_upload_is_rejected(self): + self.client.force_authenticate(user=None) + response = self.upload() + self.assertIn(response.status_code, (401, 403), response.status_code) + + def test_unauthorized_user_cannot_upload_for_a_patient(self): + self.client.force_authenticate(user=self.create_user()) + response = self.upload() + self.assertEqual(response.status_code, 403, response.data) + + def test_unauthorized_upload_persists_nothing(self): + self.client.force_authenticate(user=self.create_user()) + self.upload() + self.assertFalse(FileUpload.objects.exists()) + + +class UploadHandlerTests(MultipartUploadTestBase): + """Django's upload handlers decide memory vs temporary file.""" + + def captured_upload(self, content): + seen = {} + original = FileUpload.files_manager.__class__.put_object + + def spy(self, file_obj, file, content_type=None): + seen["type"] = type(file) + return original(self, file_obj, file, content_type=content_type) + + with patch.object(FileUpload.files_manager.__class__, "put_object", spy): + response = self.upload(content=content) + return response, seen.get("type") + + @override_settings(FILE_UPLOAD_MAX_MEMORY_SIZE=1024 * 1024) + def test_small_upload_stays_in_memory(self): + response, handler_type = self.captured_upload(jpeg_bytes((80, 80))) + self.assertEqual(response.status_code, 200, response.data) + self.assertIs(handler_type, InMemoryUploadedFile) + + @override_settings(FILE_UPLOAD_MAX_MEMORY_SIZE=1024) + def test_large_upload_is_backed_by_a_temporary_file(self): + response, handler_type = self.captured_upload(jpeg_bytes((800, 800))) + self.assertEqual(response.status_code, 200, response.data) + self.assertIs(handler_type, TemporaryUploadedFile) + + @override_settings(FILE_UPLOAD_MAX_MEMORY_SIZE=1024) + def test_temporary_file_upload_round_trips(self): + response = self.upload() + self.assertEqual(response.status_code, 200, response.data) + download = self.client.get(response.data["download_url"]) + self.assertEqual(response_content(download), self.payload_bytes) + + +class FailureConsistencyTests(MultipartUploadTestBase): + """Storage and PostgreSQL do not share a transaction; check the seam.""" + + def test_storage_failure_reports_failure(self): + with patch.object( + FileUpload.files_manager.__class__, + "put_object", + side_effect=OSError("storage down"), + ): + response = self.upload() + self.assertEqual(response.status_code, 400, response.data) + + def test_storage_failure_leaves_no_database_row(self): + with patch.object( + FileUpload.files_manager.__class__, + "put_object", + side_effect=OSError("storage down"), + ): + self.upload() + self.assertFalse(FileUpload.objects.exists()) + + def test_database_failure_rolls_the_row_back(self): + # The completion save fails after the object is written. The row is + # rolled back; the orphan object is the pre-existing B8 gap. + # + # The failure is *not* translated into "failed to upload to storage": + # storage succeeded. It propagates as a server error, because a database + # outage is not something the client can fix by changing its request. + original_save = FileUpload.save + completion_save = 2 # first save creates the row, second completes it + calls = {"n": 0} + + def failing_save(self, *args, **kwargs): + calls["n"] += 1 + if calls["n"] >= completion_save: + msg = "db down" + raise OSError(msg) + return original_save(self, *args, **kwargs) + + with patch.object(FileUpload, "save", failing_save), self.assertRaises(OSError): + self.upload() + self.assertFalse(FileUpload.objects.exists()) + + def test_multipart_upload_leaves_no_incomplete_row(self): + # cleanup_incomplete_file_uploads keys off upload_completed=False. The + # multipart path writes the object and sets the flag in one request, so + # a successful upload leaves nothing for the sweeper to collect. + response = self.upload() + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + self.assertTrue(file_obj.upload_completed) + self.assertTrue(file_obj.files_manager.exists(file_obj)) + + def test_storage_failure_is_reported_as_a_storage_failure(self): + # The other half of the split: a storage error keeps its own message and + # is still a 400, as before. + with patch.object( + FilesManager, "put_object", side_effect=OSError("bucket down") + ): + response = self.upload() + self.assertEqual(response.status_code, 400, response.data) + self.assertIn("storage", str(response.data).lower()) + self.assertFalse(FileUpload.objects.exists()) + + +IN_MEMORY = "django.core.files.storage.InMemoryStorage" + + +class ProviderNeutralTransportTests(MultipartUploadTestBase): + """ + The transport must not assume a provider (ES-02 section 33). + + Substituting a completely unrelated backend at the Django Storage boundary + is the strongest available proof: if any S3 assumption survived in the + upload path, an in-memory backend would break it. + """ + + @override_settings( + STORAGES={ + "staticfiles": {"BACKEND": IN_MEMORY}, + "patient": {"BACKEND": IN_MEMORY}, + "facility": {"BACKEND": IN_MEMORY}, + "report": {"BACKEND": IN_MEMORY}, + } + ) + def test_upload_works_against_a_substituted_backend(self): + response = self.upload() + self.assertEqual(response.status_code, 200, response.data) + file_obj = FileUpload.objects.get(external_id=response.data["id"]) + self.assertEqual(file_obj.files_manager.storage_alias, "patient") + self.assertTrue(file_obj.files_manager.exists(file_obj)) + self.assertEqual( + file_obj.files_manager.file_contents(file_obj), self.payload_bytes + ) + + @override_settings( + STORAGES={ + "staticfiles": {"BACKEND": IN_MEMORY}, + "patient": {"BACKEND": IN_MEMORY}, + "facility": {"BACKEND": IN_MEMORY}, + "report": {"BACKEND": IN_MEMORY}, + } + ) + def test_download_works_against_a_substituted_backend(self): + response = self.upload() + download = self.client.get(response.data["download_url"]) + self.assertEqual(download.status_code, 200) + self.assertEqual(response_content(download), self.payload_bytes) + + def test_transport_modules_do_not_branch_on_the_configured_provider(self): + # Inspect the AST rather than the text: these modules legitimately + # mention boto3 and the backend classes in prose, explaining what they + # replaced. What matters is that no *code* references them. + import ast + import inspect + + from care.emr.api.viewsets import file_upload as upload_module + from care.emr.utils import file_download, file_manager + + forbidden_roots = {"boto3", "botocore", "google", "storages"} + + for module in (upload_module, file_download, file_manager): + tree = ast.parse(inspect.getsource(module)) + imported = set() + attributes = set() + for node in ast.walk(tree): + if isinstance(node, ast.Import): + imported.update(alias.name.split(".")[0] for alias in node.names) + elif isinstance(node, ast.ImportFrom) and node.module: + imported.add(node.module.split(".")[0]) + elif isinstance(node, ast.Attribute): + attributes.add(node.attr) + + with self.subTest(module=module.__name__): + self.assertEqual( + imported & forbidden_roots, + set(), + f"{module.__name__} imports a provider SDK", + ) + self.assertNotIn( + "CARE_STORAGE_BACKEND", + attributes, + f"{module.__name__} branches on the configured provider", + ) + + +class UploadSchemaTests(CareAPITestBase): + """The generated OpenAPI schema must describe multipart, not JSON.""" + + def schema(self): + from drf_spectacular.generators import SchemaGenerator + + return SchemaGenerator().get_schema(request=None, public=True) + + def upload_operation(self): + return self.schema()["paths"]["/api/v1/files/upload-file/"]["post"] + + def test_request_is_multipart_only(self): + content = self.upload_operation()["requestBody"]["content"] + self.assertEqual(list(content), ["multipart/form-data"]) + + def test_file_field_is_binary(self): + schema = self.schema() + content = self.upload_operation()["requestBody"]["content"] + ref = content["multipart/form-data"]["schema"]["$ref"].split("/")[-1] + properties = schema["components"]["schemas"][ref]["properties"] + self.assertEqual(properties["file"]["type"], "string") + self.assertEqual(properties["file"]["format"], "binary") + + def test_base64_field_is_absent(self): + schema = self.schema() + content = self.upload_operation()["requestBody"]["content"] + ref = content["multipart/form-data"]["schema"]["$ref"].split("/")[-1] + properties = schema["components"]["schemas"][ref]["properties"] + self.assertNotIn("file_data", properties) + + def test_required_fields(self): + schema = self.schema() + content = self.upload_operation()["requestBody"]["content"] + ref = content["multipart/form-data"]["schema"]["$ref"].split("/")[-1] + required = schema["components"]["schemas"][ref]["required"] + self.assertIn("file", required) + self.assertNotIn("original_name", required) diff --git a/care/emr/tests/test_storage.py b/care/emr/tests/test_storage.py new file mode 100644 index 0000000000..25fb493df8 --- /dev/null +++ b/care/emr/tests/test_storage.py @@ -0,0 +1,246 @@ +""" +Storage behaviour tests (ADR-0001 / IS-01). + +Three groups: + +- object-name generation, which must stay byte-for-byte compatible; +- delegation through Django Storage, proven with an in-memory backend at the + Django Storage boundary rather than by mocking a provider SDK; +- real MinIO integration through the configured aliases. +""" + +import uuid +from types import SimpleNamespace + +from django.core.exceptions import SuspiciousFileOperation +from django.core.files.base import ContentFile +from django.core.files.storage import storages +from django.test import SimpleTestCase, override_settings + +from care.emr.models.file_upload import FileUpload +from care.emr.models.report.report_upload import ReportUpload +from care.emr.utils.file_manager import FilesManager, get_storage_name + +IN_MEMORY = "django.core.files.storage.InMemoryStorage" + + +def file_stub(file_type, internal_name): + return SimpleNamespace(file_type=file_type, internal_name=internal_name) + + +def unique_name(): + return f"is01-test-{uuid.uuid4()}" + + +class StorageNameTests(SimpleTestCase): + def test_convention_is_file_type_slash_internal_name(self): + self.assertEqual( + get_storage_name(file_stub("patient", "abc123")), "patient/abc123" + ) + + def test_each_logical_file_type(self): + for file_type in ("patient", "encounter", "consent", "diagnostic_report"): + with self.subTest(file_type=file_type): + self.assertEqual( + get_storage_name(file_stub(file_type, "obj")), f"{file_type}/obj" + ) + + def test_extension_preserved(self): + self.assertEqual( + get_storage_name(file_stub("patient", "abc.tar.gz")), "patient/abc.tar.gz" + ) + + def test_unicode_preserved(self): + name = get_storage_name(file_stub("patient", "rapport-café-日本語.pdf")) + self.assertEqual(name, "patient/rapport-café-日本語.pdf") + + def test_unusual_but_valid_names_preserved(self): + for internal_name in ("a b c.png", "file(1).pdf", "'quoted'.txt", "a+b=c.bin"): + with self.subTest(internal_name=internal_name): + self.assertEqual( + get_storage_name(file_stub("patient", internal_name)), + f"patient/{internal_name}", + ) + + def test_name_is_relative(self): + name = get_storage_name(file_stub("patient", "abc")) + self.assertFalse(name.startswith("/")) + self.assertNotIn("://", name) + self.assertNotIn("patient-bucket", name) + + def test_path_traversal_rejected(self): + traversals = [ + ("../etc", "passwd"), + ("patient", "../../etc/passwd"), + ("patient", ".."), + ("..", ".."), + ("patient", "..\\..\\windows"), + ] + for file_type, internal_name in traversals: + with ( + self.subTest(file_type=file_type, internal_name=internal_name), + self.assertRaises(SuspiciousFileOperation), + ): + get_storage_name(file_stub(file_type, internal_name)) + + def test_absolute_paths_rejected(self): + with self.assertRaises(SuspiciousFileOperation): + get_storage_name(file_stub("/patient", "abc")) + + def test_empty_components_rejected(self): + for file_type, internal_name in (("", "abc"), ("patient", ""), (None, None)): + with ( + self.subTest(file_type=file_type, internal_name=internal_name), + self.assertRaises(SuspiciousFileOperation), + ): + get_storage_name(file_stub(file_type, internal_name)) + + +class AliasMappingTests(SimpleTestCase): + def test_models_are_bound_to_logical_aliases(self): + self.assertEqual(FileUpload.files_manager.storage_alias, "patient") + self.assertEqual(ReportUpload.files_manager.storage_alias, "report") + + def test_cover_images_use_the_facility_alias(self): + from care.utils.file_uploads import cover_image + + self.assertEqual(cover_image.STORAGE_ALIAS, "facility") + + def test_report_uses_report_type_as_file_type(self): + report = ReportUpload(report_type="discharge", internal_name="abc") + self.assertEqual(get_storage_name(report), "discharge/abc") + + +@override_settings( + STORAGES={ + "staticfiles": {"BACKEND": IN_MEMORY}, + "patient": {"BACKEND": IN_MEMORY}, + "facility": {"BACKEND": IN_MEMORY}, + "report": {"BACKEND": IN_MEMORY}, + } +) +class FilesManagerDelegationTests(SimpleTestCase): + """ + Proves FilesManager delegates to whatever Django Storage backs the alias. + + No provider SDK is mocked; substituting the backend is the whole point. + """ + + def setUp(self): + self.manager = FilesManager("patient") + self.file_obj = file_stub("patient", unique_name()) + + def test_uses_the_backend_configured_for_its_alias(self): + self.assertEqual( + type(self.manager.storage), type(storages["patient"]) + ) + + def test_save_uses_the_storage_name_convention(self): + name = self.manager.put_object(self.file_obj, ContentFile(b"data")) + self.assertEqual(name, get_storage_name(self.file_obj)) + self.assertTrue(self.manager.storage.exists(name)) + + def test_content_round_trip(self): + self.manager.put_object(self.file_obj, ContentFile(b"clinical bytes")) + with self.manager.get_object(self.file_obj) as handle: + self.assertEqual(handle.read(), b"clinical bytes") + + def test_file_contents_returns_bytes(self): + self.manager.put_object(self.file_obj, ContentFile(b"payload")) + self.assertEqual(self.manager.file_contents(self.file_obj), b"payload") + + def test_exists_and_size(self): + self.assertFalse(self.manager.exists(self.file_obj)) + self.manager.put_object(self.file_obj, ContentFile(b"12345")) + self.assertTrue(self.manager.exists(self.file_obj)) + self.assertEqual(self.manager.size(self.file_obj), 5) + + def test_delete(self): + self.manager.put_object(self.file_obj, ContentFile(b"x")) + self.manager.delete_object(self.file_obj) + self.assertFalse(self.manager.exists(self.file_obj)) + + def test_deleting_a_missing_object_is_not_an_error(self): + self.manager.delete_object(self.file_obj) + + def test_put_object_returns_the_name_the_backend_chose(self): + # Overwrite-on-collision is a backend option (file_overwrite), not a + # Django Storage guarantee: InMemoryStorage renames instead. What + # FilesManager guarantees is that it reports back what the backend did. + # The overwrite behaviour CARE actually relies on is asserted against + # the configured backends in MinioIntegrationTests, and the option + # itself in care.utils.tests.test_storage_config. + first = self.manager.put_object(self.file_obj, ContentFile(b"first")) + second = self.manager.put_object(self.file_obj, ContentFile(b"second")) + self.assertEqual(first, get_storage_name(self.file_obj)) + self.assertTrue(self.manager.storage.exists(second)) + with self.manager.storage.open(second, "rb") as handle: + self.assertEqual(handle.read(), b"second") + + def test_aliases_are_independent(self): + report_manager = FilesManager("report") + self.manager.put_object(self.file_obj, ContentFile(b"patient copy")) + self.assertFalse(report_manager.exists(self.file_obj)) + + +class MinioIntegrationTests(SimpleTestCase): + """ + Mandatory local profile (IS-01 section 27.5): MinIO through S3Storage, + exercised against the running compose service via the real aliases. + + Names are UUID-based so parallel workers sharing one MinIO cannot collide. + """ + + aliases = ("patient", "facility", "report") + + def test_aliases_are_s3_storage(self): + for alias in self.aliases: + with self.subTest(alias=alias): + self.assertEqual(type(storages[alias]).__name__, "S3Storage") + + def test_round_trip_on_every_alias(self): + for alias in self.aliases: + with self.subTest(alias=alias): + storage = storages[alias] + name = f"is01-integration/{uuid.uuid4()}.txt" + payload = f"care {alias}".encode() + try: + saved = storage.save(name, ContentFile(payload)) + self.assertEqual(saved, name) + self.assertTrue(storage.exists(saved)) + self.assertEqual(storage.size(saved), len(payload)) + with storage.open(saved, "rb") as handle: + self.assertEqual(handle.read(), payload) + finally: + storage.delete(name) + self.assertFalse(storage.exists(name)) + + def test_missing_object_raises_file_not_found(self): + storage = storages["patient"] + with self.assertRaises(FileNotFoundError): + storage.open(f"is01-integration/missing-{uuid.uuid4()}", "rb") + + def test_deleting_a_missing_object_is_not_an_error(self): + storages["patient"].delete(f"is01-integration/missing-{uuid.uuid4()}") + + def test_overwrite_keeps_the_name(self): + storage = storages["patient"] + name = f"is01-integration/{uuid.uuid4()}.txt" + try: + storage.save(name, ContentFile(b"first")) + self.assertEqual(storage.save(name, ContentFile(b"second")), name) + with storage.open(name, "rb") as handle: + self.assertEqual(handle.read(), b"second") + finally: + storage.delete(name) + + def test_manager_round_trip_against_minio(self): + manager = FilesManager("patient") + file_obj = file_stub("patient", f"{uuid.uuid4()}.txt") + try: + manager.put_object(file_obj, ContentFile(b"through the manager")) + self.assertTrue(manager.exists(file_obj)) + self.assertEqual(manager.file_contents(file_obj), b"through the manager") + finally: + manager.delete_object(file_obj) + self.assertFalse(manager.exists(file_obj)) diff --git a/care/emr/tests/test_storage_transport.py b/care/emr/tests/test_storage_transport.py new file mode 100644 index 0000000000..e6c7b14e52 --- /dev/null +++ b/care/emr/tests/test_storage_transport.py @@ -0,0 +1,286 @@ +""" +Provider-neutral transport tests (ADR-0001 / ES-01). + +Asserts the property the architecture actually depends on: CARE mediates every +object transfer, so no storage-provider URL or endpoint ever reaches a client, +under either backend profile. +""" + +import io +import uuid +import warnings + +from django.core.files.base import ContentFile +from django.core.files.storage import storages +from django.core.files.uploadedfile import SimpleUploadedFile +from django.test import SimpleTestCase +from django.urls import reverse +from PIL import Image + +from care.emr.models.file_upload import FileUpload +from care.emr.utils.file_manager import FilesManager, S3FilesManager +from care.utils.tests.base import CareAPITestBase, response_content + +#: Substrings that would betray a storage-provider URL or endpoint in a response. +PROVIDER_URL_MARKERS = ( + "minio", + "amazonaws", + "s3.", + "storage.googleapis", + "googleapis.com", + "X-Amz-Signature", + "x-amz-signature", + "GoogleAccessId", + "Signature=", + ":9100", +) + + +def assert_no_provider_url(testcase, payload): + text = str(payload) + for marker in PROVIDER_URL_MARKERS: + testcase.assertNotIn(marker, text, f"provider URL marker {marker!r} leaked") + + +class NoProviderUrlInResponsesTests(CareAPITestBase): + """Nothing CARE returns may contain a storage-provider URL.""" + + def setUp(self): + super().setUp() + self.user = self.create_super_user() + self.facility = self.create_facility(user=self.user) + self.patient = self.create_patient() + + # 800x800: the cover-image validator enforces a 400x400 / 1 KB minimum. + self.file = io.BytesIO() + Image.new("RGB", (800, 800)).save(self.file, format="JPEG") + self.file.name = "scan.jpg" + self.file.seek(0) + + self.client.force_authenticate(user=self.user) + + def _upload(self): + return self.client.post( + reverse("files-upload-file"), + { + "file": SimpleUploadedFile( + "scan.jpg", self.file.getvalue(), content_type="image/jpeg" + ), + "name": "scan", + "file_type": "patient", + "file_category": "unspecified", + "associating_id": str(self.patient.external_id), + }, + format="multipart", + ) + + def test_multipart_upload_works_and_returns_no_provider_url(self): + response = self._upload() + self.assertEqual(response.status_code, 200, response.data) + assert_no_provider_url(self, response.data) + + def test_upload_response_offers_no_provider_upload_url(self): + response = self._upload() + self.assertNotIn("signed_url", response.data) + self.assertNotIn("read_signed_url", response.data) + + def test_download_url_is_a_care_route(self): + response = self._upload() + self.assertEqual( + response.data["download_url"], + reverse("files-download", args=[response.data["id"]]), + ) + + def test_detail_and_list_responses_carry_no_provider_url(self): + created = self._upload() + detail = self.client.get(reverse("files-detail", args=[created.data["id"]])) + self.assertEqual(detail.status_code, 200) + assert_no_provider_url(self, detail.data) + + listing = self.client.get( + reverse("files-list"), + {"file_type": "patient", "associating_id": str(self.patient.external_id)}, + ) + self.assertEqual(listing.status_code, 200) + assert_no_provider_url(self, listing.data) + + def test_download_through_django_returns_the_stored_bytes(self): + created = self._upload() + response = self.client.get(created.data["download_url"]) + self.assertEqual(response.status_code, 200) + self.assertEqual(response_content(response), self.file.getvalue()) + + def test_download_requires_authorization(self): + created = self._upload() + self.client.force_authenticate(user=self.create_user()) + response = self.client.get(created.data["download_url"]) + self.assertEqual(response.status_code, 403, response.content) + + def test_missing_object_yields_404_not_a_provider_error(self): + created = self._upload() + file_obj = FileUpload.objects.get(external_id=created.data["id"]) + file_obj.files_manager.delete_object(file_obj) + response = self.client.get(created.data["download_url"]) + self.assertEqual(response.status_code, 404) + + def test_facility_cover_image_url_is_a_care_route(self): + response = self.client.post( + reverse("facility-cover-image", args=[self.facility.external_id]), + {"cover_image": self.file}, + format="multipart", + ) + self.assertEqual(response.status_code, 200, response.data) + self.facility.refresh_from_db() + url = self.facility.read_cover_image_url() + self.assertEqual( + url, + reverse( + "facility-cover-image-asset", + kwargs={"external_id": self.facility.external_id}, + ), + ) + assert_no_provider_url(self, url) + + served = self.client.get(url) + self.assertEqual(served.status_code, 200) + self.assertEqual(response_content(served), self.file.getvalue()) + + def test_cover_image_is_served_without_authentication(self): + # These objects were world-readable via the bucket; CARE now serves + # them, but who can see them is unchanged. + self.client.post( + reverse("facility-cover-image", args=[self.facility.external_id]), + {"cover_image": self.file}, + format="multipart", + ) + self.facility.refresh_from_db() + url = self.facility.read_cover_image_url() + self.client.force_authenticate(user=None) + served = self.client.get(url) + self.assertEqual(served.status_code, 200) + + def test_avatar_url_is_a_care_route(self): + response = self.client.post( + reverse("users-profile-picture", args=[self.user.username]), + {"profile_picture": self.file}, + format="multipart", + ) + self.assertEqual(response.status_code, 200) + self.user.refresh_from_db() + url = self.user.read_profile_picture_url() + self.assertEqual( + url, + reverse( + "user-profile-picture-asset", kwargs={"username": self.user.username} + ), + ) + assert_no_provider_url(self, url) + self.assertEqual(self.client.get(url).status_code, 200) + + def test_asset_route_404s_when_no_image_is_set(self): + url = reverse( + "facility-cover-image-asset", + kwargs={"external_id": self.facility.external_id}, + ) + self.assertEqual(self.client.get(url).status_code, 404) + + def test_public_asset_is_cacheable(self): + # A key is minted per upload, so the bytes behind it never change and + # an intermediary cache can hold them. Without this CARE would serve + # every avatar view itself, which reading the bucket directly did not. + self.client.post( + reverse("facility-cover-image", args=[self.facility.external_id]), + {"cover_image": self.file}, + format="multipart", + ) + self.facility.refresh_from_db() + served = self.client.get(self.facility.read_cover_image_url()) + self.assertEqual(served.status_code, 200) + self.assertIn("immutable", served.headers["Cache-Control"]) + self.assertIn("public", served.headers["Cache-Control"]) + + def _create_without_uploading(self): + return self.client.post( + reverse("files-list"), + { + "name": "scan", + "original_name": "scan.jpg", + "file_type": "patient", + "file_category": "unspecified", + "associating_id": str(self.patient.external_id), + "mime_type": "image/jpeg", + }, + format="json", + ) + + def test_incomplete_upload_offers_no_download_url(self): + # The row exists before its bytes do. A download route here could only + # 404, so it is not advertised. + created = self._create_without_uploading() + self.assertEqual(created.status_code, 200, created.data) + self.assertIsNone(created.data["download_url"]) + + def test_incomplete_upload_cannot_be_downloaded(self): + created = self._create_without_uploading() + response = self.client.get(reverse("files-download", args=[created.data["id"]])) + self.assertEqual(response.status_code, 404, response.content) + + +class ProviderNeutralPersistenceTests(SimpleTestCase): + """Every alias round-trips through Django Storage under the s3 profile.""" + + aliases = ("patient", "facility", "report") + + def test_all_aliases_round_trip(self): + for alias in self.aliases: + with self.subTest(alias=alias): + storage = storages[alias] + name = f"is01-transport/{uuid.uuid4()}.bin" + try: + storage.save(name, ContentFile(b"payload")) + with storage.open(name, "rb") as handle: + self.assertEqual(handle.read(), b"payload") + finally: + storage.delete(name) + + def test_no_alias_exposes_a_signed_url_helper(self): + manager = FilesManager("patient") + for attribute in ("signed_url", "read_signed_url", "generate_presigned_url"): + self.assertFalse(hasattr(manager, attribute), attribute) + + +class PluginCompatibilityTests(SimpleTestCase): + """S3FilesManager stays importable for plugins, minus the signed URLs.""" + + def test_importable_and_deprecated(self): + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + manager = S3FilesManager("PATIENT") + self.assertTrue( + any(issubclass(w.category, DeprecationWarning) for w in caught), + "expected a DeprecationWarning", + ) + self.assertIsInstance(manager, FilesManager) + + def test_delegates_to_django_storage(self): + manager = S3FilesManager("PATIENT") + self.assertEqual(manager.storage_alias, "patient") + self.assertIs(type(manager.storage), type(storages["patient"])) + + def test_accepts_each_legacy_bucket_name(self): + for legacy, alias in ( + ("PATIENT", "patient"), + ("FACILITY", "facility"), + ("REPORT", "report"), + ): + with self.subTest(legacy=legacy): + self.assertEqual(S3FilesManager(legacy).storage_alias, alias) + + def test_rejects_unknown_alias(self): + with self.assertRaises(ValueError): + S3FilesManager("does-not-exist") + + def test_exposes_no_signed_url_methods(self): + manager = S3FilesManager("PATIENT") + for attribute in ("signed_url", "read_signed_url", "generate_presigned_url"): + self.assertFalse(hasattr(manager, attribute), attribute) diff --git a/care/emr/utils/file_download.py b/care/emr/utils/file_download.py new file mode 100644 index 0000000000..7b83b72c62 --- /dev/null +++ b/care/emr/utils/file_download.py @@ -0,0 +1,78 @@ +""" +Django-mediated object download. + +ADR-0001: a client never receives a storage-provider URL. Every read of a stored +object is served by CARE, streaming through Django Storage, so the bucket stays +private and the provider stays interchangeable. + +This module contains no provider SDK import and no provider branch. +""" + +import mimetypes + +from django.http import FileResponse +from django.urls import reverse +from rest_framework.exceptions import NotFound + +from care.emr.utils.file_manager import get_storage_name + +#: MIME types a browser may render inline; everything else downloads. +#: Preserved from the presigned ``ResponseContentDisposition`` behaviour this +#: replaces, so browser handling of patient documents is unchanged. +SAFE_INLINE_FORMATS = { + "image/jpeg", + "image/png", + "image/gif", + "image/webp", + "image/tiff", + "image/bmp", + "image/x-icon", + "application/pdf", +} + + +def storage_file_response(storage, name, *, filename, mime_type=None): + """ + Stream ``name`` from ``storage`` as an HTTP response. + + ``FileResponse`` streams in chunks and closes the handle when the response + is finished, so the object is never fully buffered in memory. + """ + if not mime_type: + mime_type = mimetypes.guess_type(filename or name)[0] + mime_type = mime_type or "application/octet-stream" + + try: + handle = storage.open(name, "rb") + except FileNotFoundError as e: + msg = "File not found in storage" + raise NotFound(msg) from e + + return FileResponse( + handle, + as_attachment=mime_type not in SAFE_INLINE_FORMATS, + filename=filename, + content_type=mime_type, + ) + + +def file_object_response(file_obj): + """Stream a ``FileUpload`` or ``ReportUpload`` through Django Storage.""" + return storage_file_response( + file_obj.files_manager.storage, + get_storage_name(file_obj), + filename=f"{file_obj.name}{file_obj.get_extension()}", + mime_type=file_obj.meta.get("mime_type"), + ) + + +def file_download_url(file_obj) -> str: + """CARE download route for a ``FileUpload``.""" + return reverse("files-download", kwargs={"external_id": file_obj.external_id}) + + +def report_download_url(report_obj) -> str: + """CARE download route for a ``ReportUpload``.""" + return reverse( + "template-reports-download", kwargs={"external_id": report_obj.external_id} + ) diff --git a/care/emr/utils/file_manager.py b/care/emr/utils/file_manager.py index 449b111b25..ce0b8cbf17 100644 --- a/care/emr/utils/file_manager.py +++ b/care/emr/utils/file_manager.py @@ -1,133 +1,171 @@ -import logging +""" +Object persistence for CARE's file models. + +ADR-0001 makes Django's Storage API the object-persistence abstraction, so this +module contains no provider SDK import and no provider branch. The provider is +selected entirely in settings; see ``config/storage.py``. -import boto3 -from botocore.exceptions import ClientError +There is no presigned-URL generation anywhere in CARE. Objects are read back +through :mod:`care.emr.utils.file_download`, which streams them from Django +Storage, so no storage-provider URL ever reaches a client. +""" -from care.utils.csp.config import get_client_config +import logging +import warnings +from typing import ClassVar + +from django.core.exceptions import SuspiciousFileOperation +from django.core.files.storage import storages logger = logging.getLogger(__name__) -SAFE_INLINE_FORMATS = { - "image/jpeg", - "image/png", - "image/gif", - "image/webp", - "image/tiff", - "image/bmp", - "image/x-icon", - "application/pdf", -} +def get_storage_name(file_obj) -> str: + """ + Return the provider-neutral object name for ``file_obj``. + The convention is ``/``, unchanged from the boto3 + key this replaces. The result is relative: it carries no bucket, no URL and + no provider endpoint. -class FileManager: + Both components are server-controlled in practice -- ``file_type`` is a + bounded choice (``FileTypeChoices`` for uploads, the report-type registry + for reports) and ``internal_name`` is generated from a UUID. The traversal + guard below is defence in depth, so that a future caller cannot quietly + write outside the intended prefix. """ - A utility class to manage all file management related operations + file_type = str(file_obj.file_type or "").strip() + internal_name = str(file_obj.internal_name or "").strip() + + if not file_type or not internal_name: + msg = "Cannot build a storage name without both file_type and internal_name" + raise SuspiciousFileOperation(msg) + + name = f"{file_type}/{internal_name}" + + # Reject anything that could escape the prefix. Names are not otherwise + # normalised: existing objects must stay addressable byte-for-byte. + if name.startswith(("/", "\\")) or any( + segment in {"..", ""} for segment in name.replace("\\", "/").split("/") + ): + msg = f"Detected path traversal attempt in storage name: {name!r}" + raise SuspiciousFileOperation(msg) + + return name + + +class FilesManager: """ + Transitional wrapper binding a CARE file model to a logical storage alias. + Retained rather than removed because ``files_manager`` is a class attribute + on ``FileUpload`` and ``ReportUpload`` and is referenced from viewsets, + tasks and report generation; rewriting every caller to use ``storages[...]`` + directly would be a wider change than IS-01 warrants. -class S3FilesManager(FileManager): - bucket_type = None + It resolves the alias, generates a provider-neutral object name and + delegates to Django Storage. It holds no provider SDK import, no provider + branch, and returns no provider response object. It is *not* the permanent + storage abstraction -- ``django.core.files.storage.storages`` is. + """ - def __init__(self, bucket_type): - self.bucket_type = bucket_type - - def signed_url(self, file_obj, duration=60 * 60, mime_type=None): - config, bucket_name = get_client_config(self.bucket_type, external=True) - s3 = boto3.client("s3", **config) - params = { - "Bucket": bucket_name, - "Key": f"{file_obj.file_type}/{file_obj.internal_name}", - } - - _mime_type = file_obj.meta.get("mime_type") or mime_type - if _mime_type: - params["ContentType"] = _mime_type - return s3.generate_presigned_url( - "put_object", - Params=params, - ExpiresIn=duration, # seconds - ) + def __init__(self, storage_alias: str): + self.storage_alias = storage_alias - def read_signed_url(self, file_obj, duration=60 * 60): - config, bucket_name = get_client_config(self.bucket_type, external=True) - s3 = boto3.client("s3", **config) + @property + def storage(self): + """The Django ``Storage`` backing this alias.""" + return storages[self.storage_alias] - mime_type = file_obj.meta.get("mime_type") - content_disposition = ( - "inline" if mime_type in SAFE_INLINE_FORMATS else "attachment" - ) + def put_object(self, file_obj, file, content_type: str | None = None) -> str: + """ + Write ``file`` and return the stored object name. - return s3.generate_presigned_url( - "get_object", - Params={ - "Bucket": bucket_name, - "Key": f"{file_obj.file_type}/{file_obj.internal_name}", - "ResponseContentDisposition": f"{content_disposition}; filename={file_obj.name}{file_obj.get_extension()}", - }, - ExpiresIn=duration, # seconds - ) + ``file`` may be any Django file or file-like object; it is passed + through without being read into memory here. - def put_object(self, file_obj, file, **kwargs): - config, bucket_name = get_client_config(self.bucket_type) - s3 = boto3.client("s3", **config) - return s3.put_object( - Body=file, - Bucket=bucket_name, - Key=f"{file_obj.file_type}/{file_obj.internal_name}", - **kwargs, - ) + ``content_type`` is an optional hint. ``S3Storage`` prefers it over the + name; ``GoogleCloudStorage`` derives the type from the name's extension + instead, so the extension carried by ``internal_name`` remains the + portable signal. + """ + name = get_storage_name(file_obj) + if content_type is not None: + file.content_type = content_type + return self.storage.save(name, file) - def get_object(self, file_obj, **kwargs): - config, bucket_name = get_client_config(self.bucket_type) - s3 = boto3.client("s3", **config) - return s3.get_object( - Bucket=bucket_name, - Key=f"{file_obj.file_type}/{file_obj.internal_name}", - **kwargs, - ) + def get_object(self, file_obj, mode: str = "rb"): + """ + Open the object and return a file-like object. - def file_contents(self, file_obj): - response = self.get_object(file_obj) - content_type = response["ContentType"] - content = response["Body"].read() - return content_type, content - - def delete_object(self, file_obj, quiet=False, **kwargs): - config, bucket_name = get_client_config(self.bucket_type) - s3 = boto3.client("s3", **config) - - try: - return s3.delete_object( - Bucket=bucket_name, - Key=f"{file_obj.file_type}/{file_obj.internal_name}", - **kwargs, - ) - except s3.exceptions.NoSuchKey as e: - if not quiet: - raise e - msg = f"Object not found: {file_obj.file_type}/{file_obj.internal_name}" - logger.debug(msg) - - def delete_objects(self, file_obj_list, quiet=False, **kwargs): - config, bucket_name = get_client_config(self.bucket_type) - s3 = boto3.client("s3", **config) - - keys = [ - f"{file_obj.file_type}/{file_obj.internal_name}" - for file_obj in file_obj_list - ] - objects = [{"Key": key} for key in keys] - - try: - return s3.delete_objects( - Bucket=bucket_name, - Delete={"Objects": objects, "Quiet": quiet}, - **kwargs, + Raises ``FileNotFoundError`` when the object does not exist. The caller + is responsible for closing it; prefer using it as a context manager. + """ + return self.storage.open(get_storage_name(file_obj), mode) + + def file_contents(self, file_obj) -> bytes: + """ + Read the whole object into memory. + + Only for callers that genuinely need the complete bytes. Prefer + :meth:`get_object` as a context manager. + """ + with self.get_object(file_obj) as file: + return file.read() + + def delete_object(self, file_obj) -> None: + """ + Delete the object. + + Deleting a missing object is not an error. That matches Django Storage + on both backends and the behaviour it replaces: S3 ``delete_object`` is + idempotent, so the previous ``NoSuchKey`` branch never actually fired. + """ + self.storage.delete(get_storage_name(file_obj)) + + def exists(self, file_obj) -> bool: + return self.storage.exists(get_storage_name(file_obj)) + + def size(self, file_obj) -> int: + return self.storage.size(get_storage_name(file_obj)) + + +class S3FilesManager(FilesManager): + """ + Deprecated. Kept only so that external plugins importing this name keep + working; see docs/xii/architecture/inventory/plugin-impact.md. + + Despite the name it is not S3-specific and never was after ADR-0001: it is + :class:`FilesManager`, delegating to whichever provider the alias is + configured for. It deliberately exposes **no** signed-URL methods -- those + were removed with the provider-specific transport. Use ``FilesManager`` or + ``django.core.files.storage.storages`` directly. + + Accepts the historical positional argument, which used to be a + ``BucketType``. Anything that is not a known alias raises, rather than + silently writing to the wrong bucket. + """ + + #: Historic BucketType.name / .value -> logical alias. + _LEGACY_ALIASES: ClassVar[dict[str, str]] = { + "PATIENT": "patient", + "FACILITY": "facility", + "REPORT": "report", + } + + def __init__(self, bucket_type): + warnings.warn( + "S3FilesManager is deprecated and will be removed; use " + "FilesManager(alias) or django.core.files.storage.storages.", + DeprecationWarning, + stacklevel=2, + ) + alias = getattr(bucket_type, "value", bucket_type) + alias = self._LEGACY_ALIASES.get(str(alias).upper(), str(alias)) + if alias not in self._LEGACY_ALIASES.values(): + msg = ( + f"Unknown storage alias {alias!r}. " + f"Expected one of: {', '.join(sorted(self._LEGACY_ALIASES.values()))}." ) - except ClientError as e: - if e.response["Error"]["Code"] == "NotImplemented": - # bulk delete is not supported by some providers: GCP - msg = f"Batch delete objects not implemented for {self.bucket_type.value} bucket" - raise NotImplementedError(msg) from e - raise + raise ValueError(msg) + super().__init__(alias) diff --git a/care/facility/models/facility.py b/care/facility/models/facility.py index a6b9460dac..a542e013e7 100644 --- a/care/facility/models/facility.py +++ b/care/facility/models/facility.py @@ -4,6 +4,7 @@ from django.core.cache import cache from django.db import models from django.db.models import IntegerChoices +from django.urls import reverse from django.utils.translation import gettext_lazy as _ from care.emr.models import FacilityOrganization @@ -205,10 +206,17 @@ class Meta: def read_cover_image_url(self): + """ + CARE route serving the cover image (ADR-0001). + + Never a storage-provider URL: the bytes are read through Django Storage + so the bucket can stay private and the provider interchangeable. + """ if self.cover_image_url: - if settings.FACILITY_CDN: - return f"{settings.FACILITY_CDN}/{self.cover_image_url}" - return f"{settings.FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT}/{settings.FACILITY_S3_BUCKET}/{self.cover_image_url}" + return reverse( + "facility-cover-image-asset", + kwargs={"external_id": self.external_id}, + ) return None def __str__(self): diff --git a/care/users/models.py b/care/users/models.py index c58695b3ad..1be8c60ab9 100644 --- a/care/users/models.py +++ b/care/users/models.py @@ -2,7 +2,6 @@ import string import uuid -from django.conf import settings from django.contrib.auth.models import AbstractUser, UserManager from django.core.validators import MaxValueValidator, MinValueValidator from django.db import models @@ -200,10 +199,17 @@ def get_cached_role_orgs(self): return data def read_profile_picture_url(self): + """ + CARE route serving the avatar (ADR-0001). + + Never a storage-provider URL: the bytes are read through Django Storage + so the bucket can stay private and the provider interchangeable. + """ if self.profile_picture_url: - if settings.FACILITY_CDN: - return f"{settings.FACILITY_CDN}/{self.profile_picture_url}" - return f"{settings.FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT}/{settings.FACILITY_S3_BUCKET}/{self.profile_picture_url}" + return reverse( + "user-profile-picture-asset", + kwargs={"username": self.username}, + ) return None def is_mfa_enabled(self): diff --git a/care/utils/csp/__init__.py b/care/utils/csp/__init__.py deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/care/utils/csp/config.py b/care/utils/csp/config.py deleted file mode 100644 index 8b20b21ae3..0000000000 --- a/care/utils/csp/config.py +++ /dev/null @@ -1,81 +0,0 @@ -import enum -from typing import TypedDict - -from django.conf import settings - - -class ClientConfig(TypedDict): - region_name: str - aws_access_key_id: str - aws_secret_access_key: str - endpoint_url: str - - -type BucketName = str - - -class CSProvider(enum.Enum): - AWS = "AWS" - AWS_ROLE_BASED = "AWS_ROLE_BASED" - GCP = "GCP" - DIGITAL_OCEAN = "DIGITAL_OCEAN" - MINIO = "MINIO" - DOCKER = "DOCKER" # localstack in docker - LOCAL = "LOCAL" # localstack on host - - -class BucketType(enum.Enum): - PATIENT = "PATIENT" - FACILITY = "FACILITY" - REPORT = "REPORT" - - -def get_facility_bucket_config(external) -> tuple[ClientConfig, BucketName]: - params = {"region_name": settings.FACILITY_S3_REGION} - if CSProvider.AWS_ROLE_BASED.value != settings.BUCKET_PROVIDER: - params["aws_access_key_id"] = settings.FACILITY_S3_KEY - params["aws_secret_access_key"] = settings.FACILITY_S3_SECRET - params["endpoint_url"] = ( - settings.FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT - if external - else settings.FACILITY_S3_BUCKET_ENDPOINT - ) - return params, settings.FACILITY_S3_BUCKET - - -def get_patient_bucket_config(external) -> tuple[ClientConfig, BucketName]: - params = {"region_name": settings.FACILITY_S3_REGION} - if CSProvider.AWS_ROLE_BASED.value != settings.BUCKET_PROVIDER: - params["aws_access_key_id"] = settings.FACILITY_S3_KEY - params["aws_secret_access_key"] = settings.FACILITY_S3_SECRET - params["endpoint_url"] = ( - settings.FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT - if external - else settings.FILE_UPLOAD_BUCKET_ENDPOINT - ) - return params, settings.FILE_UPLOAD_BUCKET - - -def get_report_bucket_config(external) -> tuple[ClientConfig, BucketName]: - """Get bucket configuration for reports - uses same bucket as patient files""" - params = {"region_name": settings.FACILITY_S3_REGION} - if CSProvider.AWS_ROLE_BASED.value != settings.BUCKET_PROVIDER: - params["aws_access_key_id"] = settings.FACILITY_S3_KEY - params["aws_secret_access_key"] = settings.FACILITY_S3_SECRET - params["endpoint_url"] = ( - settings.FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT - if external - else settings.FILE_UPLOAD_BUCKET_ENDPOINT - ) - return params, settings.FILE_UPLOAD_BUCKET - - -def get_client_config(bucket_type: BucketType, external=False): - if bucket_type == BucketType.FACILITY: - return get_facility_bucket_config(external=external) - if bucket_type == BucketType.PATIENT: - return get_patient_bucket_config(external=external) - if bucket_type == BucketType.REPORT: - return get_report_bucket_config(external=external) - msg = "Invalid Bucket Type" - raise ValueError(msg) diff --git a/care/utils/file_uploads/cover_image.py b/care/utils/file_uploads/cover_image.py index a774c451bc..198473b101 100644 --- a/care/utils/file_uploads/cover_image.py +++ b/care/utils/file_uploads/cover_image.py @@ -2,21 +2,22 @@ import secrets from typing import Literal -import boto3 -from django.conf import settings +from django.core.files.storage import storages from django.core.files.uploadedfile import UploadedFile -from care.utils.csp.config import BucketType, get_client_config - logger = logging.getLogger(__name__) +#: Cover images and avatars keep their own key convention, +#: ``/_.``, which is unrelated to the +#: ``/`` convention used by FileUpload/ReportUpload. +#: They are therefore addressed through the alias directly rather than through +#: FilesManager. +STORAGE_ALIAS = "facility" -def delete_cover_image(image_key: str, folder: Literal["cover_images", "avatars"]): - config, bucket_name = get_client_config(BucketType.FACILITY) - s3 = boto3.client("s3", **config) +def delete_cover_image(image_key: str, folder: Literal["cover_images", "avatars"]): try: - s3.delete_object(Bucket=bucket_name, Key=image_key) + storages[STORAGE_ALIAS].delete(image_key) except Exception: logger.warning("Failed to delete cover image %s", image_key) @@ -27,12 +28,11 @@ def upload_cover_image( folder: Literal["cover_images", "avatars"], old_key: str | None = None, ) -> str: - config, bucket_name = get_client_config(BucketType.FACILITY) - s3 = boto3.client("s3", **config) + storage = storages[STORAGE_ALIAS] if old_key: try: - s3.delete_object(Bucket=bucket_name, Key=old_key) + storage.delete(old_key) except Exception: logger.warning("Failed to delete old cover image %s", old_key) @@ -41,13 +41,7 @@ def upload_cover_image( f"{folder}/{object_external_id}_{secrets.token_hex(8)}.{image_extension}" ) - boto_params = { - "Bucket": bucket_name, - "Key": image_key, - "Body": image.file, - } - if settings.BUCKET_HAS_FINE_ACL: - boto_params["ACL"] = "public-read" - s3.put_object(**boto_params) - - return image_key + # No ACL is set: the object is private and served by CARE through the + # public asset routes (ADR-0001). The uploaded file is passed through + # rather than read into memory. + return storage.save(image_key, image) diff --git a/care/utils/tests/base.py b/care/utils/tests/base.py index 87204d9b18..8a53fca051 100644 --- a/care/utils/tests/base.py +++ b/care/utils/tests/base.py @@ -19,6 +19,16 @@ sys.modules["care.emr.utils.valueset_coding_type"].validate_valueset = lambda f, s, c: c +def response_content(response) -> bytes: + """ + Collect the body of a streaming response. + + Object downloads are served with ``FileResponse``, which streams in chunks + and exposes no ``.content``. + """ + return b"".join(response.streaming_content) + + class CareAPITestBase(APITestCase): fake = Faker() diff --git a/care/utils/tests/test_storage_config.py b/care/utils/tests/test_storage_config.py new file mode 100644 index 0000000000..1f61ed8891 --- /dev/null +++ b/care/utils/tests/test_storage_config.py @@ -0,0 +1,160 @@ +""" +Storage configuration tests (ADR-0001 / IS-01). + +These cover how the logical aliases are *constructed*. They never contact a +provider: the GCS cases exercise settings construction only and require no +Google credentials. +""" + +from django.conf import settings +from django.core.exceptions import ImproperlyConfigured +from django.test import SimpleTestCase + +from config.storage import ( + GCS_BACKEND, + S3_BACKEND, + SUPPORTED_STORAGE_BACKENDS, + build_object_storage, + validate_storage_backend, +) + +OBJECT_STORAGE_ALIASES = ("patient", "facility", "report") + + +class StorageBackendValidationTests(SimpleTestCase): + def test_supported_backends(self): + self.assertEqual(SUPPORTED_STORAGE_BACKENDS, ("s3", "gcs")) + + def test_valid_backends_accepted(self): + for backend in SUPPORTED_STORAGE_BACKENDS: + with self.subTest(backend=backend): + self.assertEqual(validate_storage_backend(backend), backend) + + def test_invalid_backend_rejected_and_lists_supported_values(self): + with self.assertRaises(ImproperlyConfigured) as ctx: + validate_storage_backend("azure") + message = str(ctx.exception) + self.assertIn("azure", message) + for backend in SUPPORTED_STORAGE_BACKENDS: + self.assertIn(backend, message) + + def test_build_rejects_invalid_backend(self): + with self.assertRaises(ImproperlyConfigured): + build_object_storage("nope", "some-bucket") + + +class ActiveStorageSettingsTests(SimpleTestCase): + """The aliases as actually configured for this test run.""" + + def test_default_backend_is_s3(self): + self.assertEqual(settings.CARE_STORAGE_BACKEND, "s3") + + def test_object_aliases_are_configured(self): + for alias in OBJECT_STORAGE_ALIASES: + with self.subTest(alias=alias): + self.assertIn(alias, settings.STORAGES) + self.assertEqual(settings.STORAGES[alias]["BACKEND"], S3_BACKEND) + + def test_staticfiles_remains_whitenoise(self): + self.assertEqual( + settings.STORAGES["staticfiles"]["BACKEND"], + "whitenoise.storage.CompressedManifestStaticFilesStorage", + ) + + def test_aliases_carry_a_bucket(self): + for alias in OBJECT_STORAGE_ALIASES: + with self.subTest(alias=alias): + self.assertTrue(settings.STORAGES[alias]["OPTIONS"]["bucket_name"]) + + def test_report_shares_the_patient_bucket_but_stays_a_distinct_alias(self): + patient = settings.STORAGES["patient"]["OPTIONS"]["bucket_name"] + report = settings.STORAGES["report"]["OPTIONS"]["bucket_name"] + self.assertEqual(patient, report) + self.assertIsNot(settings.STORAGES["patient"], settings.STORAGES["report"]) + + def test_alias_names_carry_no_provider_name(self): + for alias in settings.STORAGES: + with self.subTest(alias=alias): + for provider in ("s3", "gcs", "minio", "aws", "google"): + self.assertNotIn(provider, alias.lower()) + + +class S3ProfileConstructionTests(SimpleTestCase): + def test_s3_alias_uses_s3storage(self): + config = build_object_storage("s3", "patient-bucket") + self.assertEqual(config["BACKEND"], S3_BACKEND) + self.assertEqual(config["OPTIONS"]["bucket_name"], "patient-bucket") + + def test_file_overwrite_enabled(self): + # CARE generates unique internal names and the boto3 put_object this + # replaces overwrote unconditionally, so Django must not rename. + config = build_object_storage("s3", "b") + self.assertTrue(config["OPTIONS"]["file_overwrite"]) + + def test_always_private(self): + # CARE serves every object, so no alias is ever public (ADR-0001). + config = build_object_storage("s3", "b") + self.assertIsNone(config["OPTIONS"]["default_acl"]) + + def test_no_alias_can_be_made_public(self): + for alias in OBJECT_STORAGE_ALIASES: + with self.subTest(alias=alias): + self.assertIsNone(settings.STORAGES[alias]["OPTIONS"]["default_acl"]) + + def test_endpoint_omitted_for_aws(self): + # Generic AWS S3 has no custom endpoint. + config = build_object_storage("s3", "b", region_name="ap-south-1") + self.assertNotIn("endpoint_url", config["OPTIONS"]) + + def test_endpoint_included_for_s3_compatible(self): + config = build_object_storage("s3", "b", endpoint_url="http://minio:9000") + self.assertEqual(config["OPTIONS"]["endpoint_url"], "http://minio:9000") + + def test_credentials_omitted_when_absent(self): + # Role-based AWS credentials: boto3 resolves them itself. + config = build_object_storage("s3", "b", access_key=None, secret_key=None) + self.assertNotIn("access_key", config["OPTIONS"]) + self.assertNotIn("secret_key", config["OPTIONS"]) + + def test_gcs_variables_are_not_required_under_s3(self): + config = build_object_storage("s3", "b", project_id=None) + self.assertNotIn("project_id", config["OPTIONS"]) + + +class GCSProfileConstructionTests(SimpleTestCase): + """Construction only -- no Google credentials and no network access.""" + + def test_gcs_alias_uses_google_cloud_storage(self): + config = build_object_storage("gcs", "patient-bucket") + self.assertEqual(config["BACKEND"], GCS_BACKEND) + self.assertEqual(config["OPTIONS"]["bucket_name"], "patient-bucket") + + def test_s3_options_are_not_leaked_into_gcs(self): + config = build_object_storage( + "gcs", + "b", + region_name="ap-south-1", + access_key="key", + secret_key="secret", + endpoint_url="http://minio:9000", + ) + for option in ("region_name", "access_key", "secret_key", "endpoint_url"): + self.assertNotIn(option, config["OPTIONS"]) + + def test_gcs_never_uses_object_acls(self): + # Uniform bucket-level access rejects per-object ACLs. + config = build_object_storage("gcs", "b") + self.assertIsNone(config["OPTIONS"]["default_acl"]) + + def test_project_id_optional(self): + without = build_object_storage("gcs", "b") + self.assertNotIn("project_id", without["OPTIONS"]) + with_project = build_object_storage("gcs", "b", project_id="care-project") + self.assertEqual(with_project["OPTIONS"]["project_id"], "care-project") + + def test_gcs_backend_is_importable_and_constructible_without_credentials(self): + from django.utils.module_loading import import_string + + backend = import_string(GCS_BACKEND) + storage = backend(**build_object_storage("gcs", "care-bucket")["OPTIONS"]) + self.assertEqual(storage.bucket_name, "care-bucket") diff --git a/config/api_router.py b/config/api_router.py index 7f83afff0d..0986648c58 100644 --- a/config/api_router.py +++ b/config/api_router.py @@ -33,6 +33,10 @@ FacilityOrganizationUsersViewSet, FacilityOrganizationViewSet, ) +from care.emr.api.viewsets.file_assets import ( + FacilityCoverImageView, + UserProfilePictureView, +) from care.emr.api.viewsets.file_upload import FileUploadViewSet from care.emr.api.viewsets.form_submission import FormSubmissionViewSet from care.emr.api.viewsets.healthcare_service import HealthcareServiceViewSet @@ -502,6 +506,19 @@ router.register("extensions", ExtensionsViewSet, basename="extensions") app_name = "api" urlpatterns = [ + # Public image delivery (ADR-0001): CARE serves the bytes so that no + # storage-provider URL is handed to a client. Unauthenticated, matching the + # world-readable bucket objects these replace. + path( + "assets/facility//cover_image/", + FacilityCoverImageView.as_view(), + name="facility-cover-image-asset", + ), + path( + "assets/user//profile_picture/", + UserProfilePictureView.as_view(), + name="user-profile-picture-asset", + ), path("", include(router.urls)), path("", include(user_nested_router.urls)), path("", include(facility_nested_router.urls)), diff --git a/config/settings/base.py b/config/settings/base.py index 8bc59d1755..d0305edb9b 100644 --- a/config/settings/base.py +++ b/config/settings/base.py @@ -15,7 +15,11 @@ from healthy_django.healthcheck.django_cache import DjangoCacheHealthCheck from healthy_django.healthcheck.django_database import DjangoDatabaseHealthCheck -from care.utils.csp import config as csp_config +from config.storage import ( + AWS_ROLE_BASED_BUCKET_PROVIDER, + build_object_storage, + validate_storage_backend, +) from plug_config import manager from .config import * # noqa F403 @@ -522,16 +526,13 @@ # ------------------------------------------------------------------------------ +# Credential source. AWS_ROLE_BASED omits the key/secret so that boto3 resolves +# instance-role credentials itself. Every other value supplies them explicitly. BUCKET_PROVIDER = env("BUCKET_PROVIDER", default="aws").upper() BUCKET_REGION = env("BUCKET_REGION", default="ap-south-1") BUCKET_KEY = env("BUCKET_KEY", default="") BUCKET_SECRET = env("BUCKET_SECRET", default="") BUCKET_ENDPOINT = env("BUCKET_ENDPOINT", default="") -BUCKET_EXTERNAL_ENDPOINT = env("BUCKET_EXTERNAL_ENDPOINT", default=BUCKET_ENDPOINT) -BUCKET_HAS_FINE_ACL = env.bool("BUCKET_HAS_FINE_ACL", default=False) - -if BUCKET_PROVIDER not in csp_config.CSProvider.__members__: - logger.error("invalid CSP found: %s", BUCKET_PROVIDER) FILE_UPLOAD_BUCKET = env("FILE_UPLOAD_BUCKET", default="") FILE_UPLOAD_REGION = env("FILE_UPLOAD_REGION", default=BUCKET_REGION) @@ -540,12 +541,6 @@ FILE_UPLOAD_BUCKET_ENDPOINT = env( "FILE_UPLOAD_BUCKET_ENDPOINT", default=BUCKET_ENDPOINT ) -FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT = env( - "FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT", - default=( - BUCKET_EXTERNAL_ENDPOINT if BUCKET_ENDPOINT else FILE_UPLOAD_BUCKET_ENDPOINT - ), -) ALLOWED_MIME_TYPES = set( env.list( @@ -664,13 +659,76 @@ FACILITY_S3_BUCKET_ENDPOINT = env( "FACILITY_S3_BUCKET_ENDPOINT", default=BUCKET_ENDPOINT ) -FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT = env( - "FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT", - default=( - BUCKET_EXTERNAL_ENDPOINT if BUCKET_ENDPOINT else FACILITY_S3_BUCKET_ENDPOINT - ), + + +# Portable object storage (ADR-0001) +# ------------------------------------------------------------------------------ +# Provider selection is configuration-only. Application code addresses the +# logical aliases below and never learns which provider implements them. +CARE_STORAGE_BACKEND = validate_storage_backend( + env("CARE_STORAGE_BACKEND", default="s3").strip().lower() +) + +# Provider-neutral bucket names. Each falls back to the pre-existing setting so +# that no local or deployed configuration has to be rewritten. `report` shares +# the patient bucket, matching the behaviour it replaces, but stays a separate +# alias so it can be pointed elsewhere without touching application code. +CARE_PATIENT_STORAGE_BUCKET = env( + "CARE_PATIENT_STORAGE_BUCKET", default=FILE_UPLOAD_BUCKET ) -FACILITY_CDN = env("FACILITY_CDN", default=None) +CARE_FACILITY_STORAGE_BUCKET = env( + "CARE_FACILITY_STORAGE_BUCKET", default=FACILITY_S3_BUCKET +) +CARE_REPORT_STORAGE_BUCKET = env( + "CARE_REPORT_STORAGE_BUCKET", default=FILE_UPLOAD_BUCKET +) + +# Optional; Application Default Credentials are used when unset. Named per +# 07-configuration-reference.md section 12.4. +GCS_PROJECT_ID = env("GCS_PROJECT_ID", default="") + +# Role-based AWS credentials: omit key/secret so the SDK resolves the instance +# role itself. The endpoint is suppressed along with them, preserving the +# behaviour of the boto3 client this replaces -- a role-based deployment always +# talked to the default AWS endpoint. A deployment that needs an instance role +# *and* a custom endpoint (a VPC endpoint, or an S3-compatible gateway) is +# therefore not expressible today; splitting the two would be a config change, +# not a refactor, so it is left for whoever first needs it. +_ROLE_BASED_BUCKET = AWS_ROLE_BASED_BUCKET_PROVIDER == BUCKET_PROVIDER + +STORAGES = { + # staticfiles is configured near the top of this file and stays on + # WhiteNoise; only the object-storage aliases are added here because they + # depend on the bucket settings defined above. + **STORAGES, + "patient": build_object_storage( + CARE_STORAGE_BACKEND, + CARE_PATIENT_STORAGE_BUCKET, + region_name=FILE_UPLOAD_REGION, + access_key=None if _ROLE_BASED_BUCKET else FILE_UPLOAD_KEY, + secret_key=None if _ROLE_BASED_BUCKET else FILE_UPLOAD_SECRET, + endpoint_url=None if _ROLE_BASED_BUCKET else FILE_UPLOAD_BUCKET_ENDPOINT, + project_id=GCS_PROJECT_ID, + ), + "facility": build_object_storage( + CARE_STORAGE_BACKEND, + CARE_FACILITY_STORAGE_BUCKET, + region_name=FACILITY_S3_REGION, + access_key=None if _ROLE_BASED_BUCKET else FACILITY_S3_KEY, + secret_key=None if _ROLE_BASED_BUCKET else FACILITY_S3_SECRET, + endpoint_url=None if _ROLE_BASED_BUCKET else FACILITY_S3_BUCKET_ENDPOINT, + project_id=GCS_PROJECT_ID, + ), + "report": build_object_storage( + CARE_STORAGE_BACKEND, + CARE_REPORT_STORAGE_BUCKET, + region_name=FILE_UPLOAD_REGION, + access_key=None if _ROLE_BASED_BUCKET else FILE_UPLOAD_KEY, + secret_key=None if _ROLE_BASED_BUCKET else FILE_UPLOAD_SECRET, + endpoint_url=None if _ROLE_BASED_BUCKET else FILE_UPLOAD_BUCKET_ENDPOINT, + project_id=GCS_PROJECT_ID, + ), +} # current hosted domain CURRENT_DOMAIN = env("CURRENT_DOMAIN", default="localhost:4000") diff --git a/config/settings/config.py b/config/settings/config.py index dcca022d37..467ff62854 100644 --- a/config/settings/config.py +++ b/config/settings/config.py @@ -18,9 +18,27 @@ MAX_FAVORITES_PER_LIST = env.int("MAX_FAVORITES_PER_LIST", default=50) -# Maximum file upload size in MB +# Maximum file upload size in MB. Enforced by CARE against UploadedFile.size +# before anything is written to storage. MAX_FILE_UPLOAD_SIZE = env.int("MAX_FILE_UPLOAD_SIZE", default=5) +# Upload transport limits (ADR-0002). Stated explicitly rather than left to +# Django's implicit defaults, which is what these values are. +# +# An upload larger than FILE_UPLOAD_MAX_MEMORY_SIZE is spooled to a +# TemporaryUploadedFile by Django's upload handlers instead of being held in +# memory. At the defaults below, anything over 2.5 MB is temp-file backed while +# MAX_FILE_UPLOAD_SIZE still caps the total at 5 MB. +# +# DATA_UPLOAD_MAX_MEMORY_SIZE bounds the non-file part of a request body. File +# parts of a multipart request are exempt, so it does not cap upload size. +FILE_UPLOAD_MAX_MEMORY_SIZE = env.int( + "FILE_UPLOAD_MAX_MEMORY_SIZE", default=2621440 +) # 2.5 MB +DATA_UPLOAD_MAX_MEMORY_SIZE = env.int( + "DATA_UPLOAD_MAX_MEMORY_SIZE", default=2621440 +) # 2.5 MB + LOCATION_MAX_DEPTH = env.int("LOCATION_MAX_DEPTH", default=10) ORGANIZATION_MAX_DEPTH = env.int("ORGANIZATION_MAX_DEPTH", default=10) diff --git a/config/storage.py b/config/storage.py new file mode 100644 index 0000000000..66df692b26 --- /dev/null +++ b/config/storage.py @@ -0,0 +1,101 @@ +""" +Construction of CARE's logical object-storage aliases. + +ADR-0001 makes Django's Storage API the object-persistence abstraction and +delegates provider implementations to ``django-storages``. Provider selection is +a configuration concern only: application code addresses the logical aliases +``patient``, ``facility`` and ``report`` and never learns which provider serves +them. + +This module is settings-level configuration code. It is deliberately *not* an +application storage framework -- application code must never import it and must +use ``django.core.files.storage.storages`` instead. +""" + +from django.core.exceptions import ImproperlyConfigured + +S3_BACKEND = "storages.backends.s3.S3Storage" +GCS_BACKEND = "storages.backends.gcloud.GoogleCloudStorage" + +#: Values accepted by the ``CARE_STORAGE_BACKEND`` setting. +SUPPORTED_STORAGE_BACKENDS = ("s3", "gcs") + +#: ``BUCKET_PROVIDER`` value meaning "resolve AWS credentials from the instance +#: role" -- key and secret are then omitted entirely. +AWS_ROLE_BASED_BUCKET_PROVIDER = "AWS_ROLE_BASED" + + +def validate_storage_backend(backend: str) -> str: + """Return ``backend`` if supported, otherwise raise ``ImproperlyConfigured``.""" + if backend not in SUPPORTED_STORAGE_BACKENDS: + supported = ", ".join(SUPPORTED_STORAGE_BACKENDS) + msg = ( + f"Invalid CARE_STORAGE_BACKEND: {backend!r}. " + f"Supported values are: {supported}." + ) + raise ImproperlyConfigured(msg) + return backend + + +def build_object_storage( + backend: str, + bucket_name: str, + *, + region_name: str | None = None, + access_key: str | None = None, + secret_key: str | None = None, + endpoint_url: str | None = None, + project_id: str | None = None, +) -> dict: + """ + Build a single ``STORAGES`` entry for one logical alias. + + Only options that carry a value are emitted, so that: + + - AWS S3 works without ``endpoint_url``; + - role-based AWS credentials work when no key/secret is supplied; + - Google Cloud Storage uses Application Default Credentials. + + ``file_overwrite`` is set explicitly rather than left to the backend + default. CARE generates a unique ``internal_name`` per object and the + behaviour it replaces (``boto3.put_object``) overwrites unconditionally, so + Django must not rename on collision. + + No alias is ever public. Every object is served by CARE through Django + Storage (ADR-0001), so buckets can and should be private. + + The S3 arguments -- ``region_name``, ``access_key``, ``secret_key`` and + ``endpoint_url`` -- have no Google Cloud Storage equivalent and are ignored + when ``backend`` is ``"gcs"``. Callers pass the full set unconditionally, so + switching a deployment to GCS silently stops using whatever those variables + held; GCS authenticates through Application Default Credentials instead. + """ + validate_storage_backend(backend) + + if backend == "gcs": + # region/key/secret/endpoint are deliberately dropped here: see docstring. + options: dict = { + "bucket_name": bucket_name, + "file_overwrite": True, + # Uniform bucket-level access: never per-object ACLs (ADR-0001). + "default_acl": None, + } + if project_id: + options["project_id"] = project_id + return {"BACKEND": GCS_BACKEND, "OPTIONS": options} + + options = { + "bucket_name": bucket_name, + "file_overwrite": True, + # Private: CARE serves every object, so no public ACL is ever needed. + "default_acl": None, + } + if region_name: + options["region_name"] = region_name + if access_key: + options["access_key"] = access_key + if secret_key: + options["secret_key"] = secret_key + if endpoint_url: + options["endpoint_url"] = endpoint_url + return {"BACKEND": S3_BACKEND, "OPTIONS": options} diff --git a/docs/xii/adr/ADR-0001-django-storage.md b/docs/xii/adr/ADR-0001-django-storage.md new file mode 100644 index 0000000000..41033e3c03 --- /dev/null +++ b/docs/xii/adr/ADR-0001-django-storage.md @@ -0,0 +1,363 @@ +# ADR-0001: Portable Object Storage using Django Storage API + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE GCP Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +--- + +# Context + +CARE currently implements object persistence through provider-specific code built +around MinIO and Amazon S3 semantics. + +The current implementation directly depends on provider SDKs (primarily `boto3`) +through custom storage management classes and helper functions. + +This approach has several disadvantages: + +- the application is coupled to one provider implementation; +- object persistence logic is mixed with provider-specific behavior; +- switching storage providers requires application changes; +- introducing new providers requires additional storage implementations; +- testing requires mocking provider SDKs instead of using Django abstractions; +- ordinary CRUD operations bypass Django's storage interface. + +The project intends to support several deployment profiles: + +- local development; +- self-hosted deployments; +- generic S3-compatible object storage; +- Google Cloud Platform; +- future cloud providers where appropriate. + +The application itself should not require modifications when changing object +storage providers. + +--- + +# Decision + +CARE SHALL adopt Django's Storage API as the single application-level abstraction +for object persistence. + +Provider-specific SDKs SHALL NOT be used directly by application code for normal +object persistence operations. + +Provider implementations SHALL be delegated to `django-storages`. + +The resulting architecture is: + +CARE + +↓ + +Django Storage API + +↓ + +django-storages + +↓ + +Provider backend + +Where the provider backend may be: + +- S3Storage +- GoogleCloudStorage +- another Django-compatible backend in the future + +--- + +# Local profile + +The default local development profile SHALL continue using MinIO. + +MinIO SHALL be accessed through: + +```text +storages.backends.s3.S3Storage +``` + +The local Docker environment SHALL continue working without: + +- Google Cloud credentials; +- service-account JSON files; +- GCP projects; +- internet connectivity to GCP. + +The default backend SHALL therefore remain S3-compatible. + +--- + +# Cloud profile + +Google Cloud Storage SHALL be supported as the initial managed-cloud deployment +profile. + +Google Cloud Storage is **not** the architecture. + +Google Cloud Storage is one implementation of the storage abstraction. + +Switching between MinIO and Google Cloud Storage SHALL require configuration +changes only. + +Application code SHALL remain unchanged. + +--- + +# Logical storage aliases + +Application code SHALL refer only to logical storage aliases. + +Initial aliases are: + +- patient +- facility +- report +- staticfiles + +Application code SHALL NOT know which provider implements an alias. + +Aliases represent business intent rather than infrastructure. + +--- + +# Configuration + +Storage provider selection SHALL be configuration-driven. + +A setting similar to: + +```text +CARE_STORAGE_BACKEND +``` + +SHALL determine which provider implementation is used. + +Initial supported values are expected to include: + +- s3 +- gcs + +The default SHALL remain: + +```text +s3 +``` + +No provider-specific branching SHALL exist in application logic. + +Provider selection belongs entirely in Django settings. + +--- + +# Object persistence + +Ordinary persistence operations SHALL use Django Storage methods such as: + +- save +- open +- exists +- delete +- size + +Application code SHALL NOT instantiate: + +- boto3 clients +- Google Cloud Storage clients +- provider-specific CRUD helpers + +for ordinary persistence. + +--- + +# File transport + +This ADR is about object persistence. It does not define the HTTP transport +layer, which is ADR-0002's subject. + +**Revised 2026-08-07.** As originally written this section said the upload and +download APIs would remain unchanged until a later transport phase. That did not +survive contact with the decision itself: presigned URLs are provider-specific +by construction, so leaving them in place would have left a provider seam in the +one place this ADR set out to remove it, and would have kept every bucket +public. The IS-01 completion pass therefore removed them. + +Removed by IS-01: + +- browser presigned uploads; +- browser presigned downloads; +- the unsigned bucket URLs serving cover images and avatars. + +Objects are now read back through CARE, which authorizes each request and +streams the bytes through Django Storage. Every bucket can be private. + +Left to ADR-0002, and since delivered by ES-02: + +- **base64 uploads.** `POST /api/v1/files/upload-file/` accepted a base64 body + and buffered the decoded file in memory. Replacing it with + `multipart/form-data` was a transport-performance change that did not affect + provider portability, which is why it belonged to ADR-0002 rather than here. + It is now multipart; no base64 upload path remains. + +--- + +# Static files + +Static assets SHALL continue using Django's existing staticfiles backend. + +This ADR applies only to application object storage. + +--- + +# Compatibility + +The implementation SHOULD preserve compatibility with: + +- MinIO +- AWS S3 +- S3-compatible providers supported by django-storages +- Google Cloud Storage + +No implementation shall assume a specific provider unless required by provider +configuration. + +--- + +# Consequences + +## Positive + +- Provider portability. +- Simpler testing. +- Better alignment with Django architecture. +- Smaller maintenance surface. +- Easier future provider additions. +- Cleaner separation between application and infrastructure. +- Reduced provider-specific code. + +## Negative + +- Some provider-specific optimizations may require redesign. +- Existing storage helpers require refactoring. +- Temporary compatibility code may be required during migration. + +--- + +# Alternatives Considered + +## Keep the existing boto3-based implementation + +Rejected. + +It increases provider coupling and duplicates functionality already provided by +Django. + +--- + +## Create a custom storage abstraction + +Rejected. + +Django already defines a mature storage abstraction. + +Introducing another abstraction would duplicate framework functionality and +increase maintenance cost. + +--- + +## Implement separate managers for each provider + +Examples: + +- S3FilesManager +- GCSFilesManager +- AzureFilesManager + +Rejected. + +This approach scales poorly and encourages provider-specific application logic. + +--- + +## Adopt Google Cloud Storage directly + +Rejected. + +The objective is provider portability. + +Google Cloud Storage is an implementation target, not the architectural +abstraction. + +--- + +# Out of Scope + +This ADR does not define: + +- multipart uploads; +- frontend upload APIs; +- signed URL removal; +- Cloud Tasks; +- Celery replacement; +- Redis replacement; +- PostgreSQL cache; +- Terraform; +- deployment automation. + +These subjects are addressed by separate ADRs and implementation +specifications. + +--- + +# Related Documents + +- Architecture 00–07 +- Runtime Inventory +- Storage Inventory +- IS-01 Storage Modernization +- ADR-0002: Server-Mediated File Transport +- ADR-0003: Configurable Asynchronous Execution +- Future ADR: Cache and Distributed Locks + +--- + +# Implementation Status + +- [x] Decision accepted. +- [x] IS-01 completed. *(2026-08-07)* +- [x] IS-02 completed. *(2026-08-07)* *Delivered as ES-02 under ADR-0002.* +- [x] Legacy storage removed. *`S3FilesManager` survives only as a deprecated + plugin shim delegating to Django Storage; `care/utils/csp/` is deleted.* +- [x] Legacy signed URL flows removed. *No application code generates a + storage-provider URL. Objects are served by CARE through Django Storage.* + +## What IS-01 delivered + +- Django Storage API is the single object-persistence abstraction; provider + implementations come from `django-storages`. +- `CARE_STORAGE_BACKEND` selects `s3` (default) or `gcs`. Switching is a + configuration change only — no application code path differs. +- Logical aliases `patient`, `facility` and `report`; `staticfiles` unchanged on + WhiteNoise. +- All object transport is mediated by CARE. Presigned upload and download are + gone, along with the unsigned bucket URLs that served cover images and + avatars. Every bucket can now be private. + +## What IS-02 delivered + +The base64 upload transport at `POST /api/v1/files/upload-file/` has been +replaced with `multipart/form-data` and Django upload handlers, under ADR-0002. +That was transport performance, not provider portability, so it did not affect +this decision — the persistence seam established here is unchanged, and the +multipart path hands its `UploadedFile` to the same `Storage.save()`. + +One item outside both specs remains open: `care/emr/tasks/report_generation.py` +retries on `botocore`'s `ClientError`, which cannot fire under `gcs`. It +constructs no client and performs no storage operation, so it is not a breach of +this decision, but it does mean report generation is not yet production-ready on +GCS. See `02-target-runtime.md` §11 and `unresolved-items.md` S2. diff --git a/docs/xii/adr/ADR-0002-file-transport.md b/docs/xii/adr/ADR-0002-file-transport.md new file mode 100644 index 0000000000..a597c6335d --- /dev/null +++ b/docs/xii/adr/ADR-0002-file-transport.md @@ -0,0 +1,536 @@ +# ADR-0002: Server-Mediated File Transport + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +--- + +## Context + +CARE historically supported multiple file-transfer mechanisms. + +The verified runtime included: + +- direct browser-to-object-storage uploads through presigned URLs; +- direct downloads through provider-generated or provider-facing URLs; +- a Django-proxied upload endpoint that transmitted complete file contents as base64; +- provider-specific transport behavior coupled to S3/MinIO concepts. + +ADR-0001 established Django Storage API as the single application-level abstraction for object persistence. + +After completion of ES-01: + +- ordinary object persistence uses Django Storage API; +- MinIO is supported locally through `S3Storage`; +- generic S3-compatible storage remains supported; +- GCS is supported through `GoogleCloudStorage`; +- signed upload URLs have been removed; +- signed download URLs have been removed; +- downloads are mediated by CARE; +- provider-specific bucket URLs are no longer part of the client contract. + +The remaining transport issue is the upload representation. + +The current Django-mediated upload path still transports complete file contents encoded as base64. + +Base64 file transport has several disadvantages: + +- approximately 33% representation overhead before additional JSON overhead; +- complete file contents tend to be materialized in memory; +- normal browser and HTTP file-upload facilities are bypassed; +- Django upload handlers cannot be used naturally; +- large-file handling becomes less efficient; +- API schemas represent binary content as application JSON instead of file content. + +The application should expose one provider-independent HTTP file contract regardless of which object-storage backend is configured. + +--- + +## Decision + +All supported file uploads and downloads SHALL be mediated by CARE, and authenticated except for the two public asset classes named under [Authorization](#public-asset-exception). + +Clients SHALL communicate only with CARE. + +CARE SHALL communicate with object storage exclusively through Django Storage API as established by ADR-0001. + +The target upload flow is: + +```text +Client + | + | multipart/form-data + v +CARE Django API + | + | UploadedFile / Django upload handlers + v +CARE validation and authorization + | + v +Django Storage API + | + v +Configured object-storage backend +``` + +The target download flow is: + +```text +Client + | + | authenticated CARE request + v +CARE Django API + | + | authorization + v +Django Storage API + | + | file-like object + v +streaming HTTP response +``` + +The public HTTP contract SHALL be independent of: + +- MinIO; +- AWS S3; +- S3-compatible providers; +- Google Cloud Storage; +- future Django-compatible storage providers. + +--- + +## Upload Transport + +File uploads SHALL use: + +```text +multipart/form-data +``` + +CARE SHALL receive uploaded content through Django's normal upload facilities. + +The implementation SHOULD use: + +- `UploadedFile`; +- `InMemoryUploadedFile`; +- `TemporaryUploadedFile`; +- Django upload handlers; +- DRF file fields where applicable. + +CARE SHALL NOT require complete file contents to be embedded inside JSON. + +The existing base64 file-content transport SHALL be removed from the target API. + +--- + +## Download Transport + +Downloads SHALL continue to pass through authenticated CARE endpoints, with the +sole exception of the two public asset classes described under +[Authorization](#public-asset-exception). File and report downloads are never +exempt. + +CARE SHALL: + +1. authenticate the caller; +2. authorize access; +3. resolve the logical storage alias; +4. open the object through Django Storage; +5. return the object using a streaming response such as `FileResponse`; +6. set appropriate content type; +7. set safe content disposition; +8. handle missing objects consistently. + +The storage provider SHALL not define the public download contract. + +--- + +## Direct Object-Storage Access + +Clients SHALL NOT: + +- request presigned upload URLs; +- request presigned download URLs; +- upload directly to MinIO; +- upload directly to AWS S3; +- upload directly to GCS; +- download directly from provider-generated storage URLs as the normal application flow; +- receive storage credentials; +- select storage buckets; +- depend on storage-provider endpoints. + +Provider-specific URLs SHALL NOT be part of the normal CARE API contract. + +--- + +## Provider Portability + +The same upload and download API SHALL work with any configured Django Storage backend supported by CARE. + +Changing: + +```text +CARE_STORAGE_BACKEND=s3 +``` + +to: + +```text +CARE_STORAGE_BACKEND=gcs +``` + +or another future supported backend SHALL NOT require frontend changes. + +Provider selection remains a server-side deployment concern. + +--- + +## Authorization + +CARE SHALL remain the authorization boundary for file access. + +Knowledge of: + +- an object name; +- a bucket name; +- a previous download path; + +SHALL NOT constitute authorization. + +Uploads and downloads SHALL continue using CARE's existing permission and domain model. + +This ADR does not redesign authorization. + +### Public asset exception + +Two asset classes are deliberately exempt from *authentication*, and only those +two: **facility cover images** and **user profile pictures**, served by +`facility-cover-image-asset` and `user-profile-picture-asset` +(`care/emr/api/viewsets/file_assets.py`). + +These objects were already world-readable directly from the bucket before +ADR-0001. Requiring authentication to read them would be a new restriction +rather than a preserved one, and the routes exist to make the *bucket* private, +not to make the images less visible. Who can see them is unchanged; what changed +is that CARE serves the bytes, so no provider URL is exposed and no object needs +a public ACL. + +Uploading or replacing either asset remains authenticated and authorized. + +Every other download — clinical files and generated reports — SHALL require an +authenticated, authorized CARE request. No further exemption SHALL be added +without amending this ADR. + +--- + +## Validation + +Uploads SHALL pass through CARE validation. + +The implementation SHALL preserve or enforce applicable rules for: + +- file presence; +- maximum file size; +- file extension; +- MIME type; +- logical file type; +- associated patient or facility; +- ownership; +- authorization; +- filename safety. + +Browser-provided MIME metadata SHALL not automatically be treated as authoritative when stronger existing validation exists. + +This ADR does not require adding heavyweight content-inspection or antivirus infrastructure. + +--- + +## Upload Memory Management + +The implementation SHALL use Django upload-handler behavior rather than base64-decoding complete request bodies. + +Small files MAY remain in memory. + +Larger supported files SHOULD be represented by temporary-file-backed uploads according to Django configuration. + +Relevant limits SHALL be explicit. + +Examples include: + +```text +CARE_MAX_UPLOAD_SIZE +FILE_UPLOAD_MAX_MEMORY_SIZE +DATA_UPLOAD_MAX_MEMORY_SIZE +``` + +Exact setting names SHALL follow the project's configuration reference. + +Temporary files are ephemeral and SHALL NOT be treated as durable storage. + +--- + +## Persistence + +After HTTP validation, file persistence SHALL continue through Django Storage API. + +Where possible, the `UploadedFile` or equivalent file-like object SHOULD be passed directly to: + +```python +storage.save(name, uploaded_file) +``` + +The transport layer SHOULD NOT perform an unnecessary complete `.read()` followed by a second in-memory representation. + +Object naming and logical storage-alias selection remain governed by ADR-0001 / ES-01. + +--- + +## Database and Storage Consistency + +PostgreSQL and object storage do not participate in a shared atomic transaction. + +The implementation SHALL therefore define failure behavior for cases such as: + +- storage write fails; +- database update fails after storage write; +- object exists but metadata creation fails; +- incomplete upload state remains. + +The implementation SHALL use the smallest correct compensation or cleanup strategy. + +This ADR does not introduce distributed transactions. + +--- + +## Response Contract + +Successful upload responses SHALL remain provider-neutral. + +They MAY expose application-level values such as: + +- CARE file identifier; +- filename; +- MIME type; +- size; +- metadata; +- CARE-relative `download_url`; +- logical processing status. + +They SHALL NOT expose: + +- bucket URL; +- storage endpoint; +- S3 URL; +- GCS URL; +- signed URL; +- storage credentials. + +--- + +## Streaming Downloads + +Normal downloads SHOULD be streamed. + +The application SHALL NOT read complete objects into memory merely to construct ordinary file responses. + +Whole-file reads may remain inside internal processing operations that explicitly require complete bytes. + +Those internal processing cases are not considered HTTP file transport. + +--- + +## HTTP Range Requests + +HTTP range support is not guaranteed by this decision. + +If media seeking or large-file requirements later require byte ranges, that capability SHALL be designed and tested explicitly. + +Range support SHALL not be assumed merely because the underlying object-storage provider supports it. + +--- + +## Static Files + +This ADR does not apply to Django static assets. + +Static files continue using the existing Django/WhiteNoise configuration established by the project. + +--- + +## Consequences + +### Positive + +- The client is completely independent of the storage provider. +- CARE remains the single authentication and authorization boundary. +- Base64 overhead is removed. +- Django upload handlers become usable. +- Large supported uploads can avoid unnecessary in-memory duplication. +- MinIO, S3 and GCS share the same HTTP contract. +- Provider changes do not require frontend changes. +- Storage credentials remain server-side. +- Validation and audit behavior remain centralized. + +### Negative + +- All file traffic passes through the CARE application runtime. +- Application bandwidth usage increases compared with direct-to-bucket transfers. +- File-transfer duration contributes to request execution time. +- Large-file support requires explicit runtime sizing and limits. +- Extremely large uploads may eventually require a different transport design. + +--- + +## Alternatives Considered + +### Preserve Base64 Uploads + +Rejected. + +Base64 is inefficient for general binary HTTP transport and prevents natural use of Django upload handlers. + +### Reintroduce Presigned Uploads + +Rejected. + +Presigned uploads expose storage-provider behavior to clients and break the selected provider-neutral application boundary. + +### Use Presigned Uploads Only for GCS or S3 + +Rejected. + +The client contract must not depend on the configured provider. + +### Use Presigned Uploads Only for Large Files + +Deferred. + +A future ADR may introduce a specialized large-object transport if actual requirements demonstrate that the server-mediated model is insufficient. + +### Build a Custom Streaming Upload Framework + +Rejected. + +Django already provides established multipart and upload-handler abstractions. + +### Use Provider-Native Multipart or Resumable Uploads + +Deferred. + +These mechanisms increase provider coupling and are unnecessary for the initial verified requirements. + +--- + +## Out of Scope + +This ADR does not define: + +- object-storage provider selection; +- storage backend implementation; +- Cloud Tasks; +- Celery migration; +- report-generation retry behavior; +- Redis; +- cache; +- distributed locks; +- rate limiting; +- Cloud Run deployment; +- Cloud SQL; +- Terraform; +- CI/CD; +- CDN architecture; +- antivirus scanning; +- resumable provider-native uploads; +- very large video ingestion. + +These concerns are governed by separate ADRs and Engineering Specifications. + +--- + +## Relationship to ADR-0001 + +ADR-0001 answers: + +> How does CARE persist and retrieve objects independently of the storage provider? + +Answer: + +```text +Django Storage API +``` + +ADR-0002 answers: + +> How do file bytes travel between clients and CARE? + +Answer: + +```text +multipart upload through CARE +streaming download through CARE +``` + +The two concerns are intentionally separate. + +--- + +## Related Documents + +- ADR-0001: Portable Object Storage using Django Storage API +- ES-01: Portable Storage Modernization +- ES-02: File Transport Modernization +- Frontend File Flow Inventory +- Storage Call-Site Inventory +- Configuration Reference +- Testing Strategy + +--- + +## Implementation Status + +Complete after ES-02 *(2026-08-07)*: + +- [x] Provider-specific signed uploads removed. *(ES-01)* +- [x] Provider-specific signed downloads removed. *(ES-01)* +- [x] Downloads mediated by CARE. *(ES-01)* +- [x] Download persistence uses Django Storage. *(ES-01)* +- [x] Provider URLs removed from the normal client contract. *(ES-01)* +- [x] Base64 upload transport removed. *`file_data` deleted with no fallback.* +- [x] Multipart upload implemented. *`POST /api/v1/files/upload-file/` accepts + `multipart/form-data` with a `file` part.* +- [x] Django upload handlers verified. *Tested at a lowered + `FILE_UPLOAD_MAX_MEMORY_SIZE`: `InMemoryUploadedFile` below the threshold, + `TemporaryUploadedFile` above it.* +- [x] Upload size limits verified/configured. *`MAX_FILE_UPLOAD_SIZE` (5 MB) + enforced against `UploadedFile.size`; `FILE_UPLOAD_MAX_MEMORY_SIZE` and + `DATA_UPLOAD_MAX_MEMORY_SIZE` stated explicitly at Django's defaults.* +- [x] Multipart API schema implemented. *`multipart/form-data` only; `file` is + `string`/`binary`; `file_data` absent.* +- [x] Upload authorization regression tests completed. *Unauthenticated and + unauthorized uploads are rejected and persist nothing.* +- [x] MinIO multipart round trip verified. *Upload through Django to `S3Storage` + to MinIO, then download back through Django.* +- [x] Provider-neutral GCS multipart behavior verified. *The flow is exercised + against a substituted backend, and the transport modules are asserted by AST + to import no provider SDK and never read `CARE_STORAGE_BACKEND`.* +- [x] ES-02 completed. + +### Notes + +**Memory.** CARE never materialises the file: size comes from +`UploadedFile.size`, MIME sniffing reads only the leading 2048 bytes, and the +`UploadedFile` is passed straight to `Storage.save()`. + +**Range requests** remain unsupported, as this ADR states. `FileResponse` may +advertise ranges for file-backed responses, but no range behaviour is designed +or tested here. + +**Not addressed by this ADR.** `POST /api/v1/files/` and `mark_upload_completed` +survive from the presigned flow and are now vestigial; nothing writes to storage +between them. Recorded in the frontend file-flow inventory §12.5. diff --git a/docs/xii/adr/ADR-0003-asynchronous-execution.md b/docs/xii/adr/ADR-0003-asynchronous-execution.md new file mode 100644 index 0000000000..a0890bb446 --- /dev/null +++ b/docs/xii/adr/ADR-0003-asynchronous-execution.md @@ -0,0 +1,324 @@ +# ADR-0003: Configurable Asynchronous Execution + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +## Context + +CARE currently uses Celery with Redis for asynchronous execution and Celery Beat for periodic scheduling. + +Repository inspection established that task decoration does not necessarily imply asynchronous execution: + +- eight task definitions were identified; +- only a limited number of call sites dispatch tasks asynchronously; +- several task-decorated functions are called synchronously; +- some periodic work is registered through Celery Beat; +- migrations and initialization commands currently execute during Celery startup. + +The project must preserve: + +- local Docker Compose behavior; +- existing Celery compatibility; +- synchronous behavior where CARE currently invokes task logic synchronously; +- portability beyond a single cloud provider. + +The initial GCP profile should avoid a permanently polling worker and use managed request-driven execution where appropriate. + +However, Cloud Tasks is a deployment implementation, not the application architecture. + +## Decision + +CARE SHALL separate reusable task behavior from task-transport and worker frameworks. + +The application SHALL support configurable asynchronous execution backends. + +The initial supported execution profiles are: + +```text +Local or traditional: +Celery + Redis + +Initial GCP profile: +Cloud Tasks + private HTTP worker + +Optional future profile: +A separately approved PostgreSQL-backed queue +``` + +Application call sites SHALL dispatch asynchronous work through a narrow application API. + +The application SHALL not treat Celery as the definition of task behavior. + +## Reusable task behavior + +Task logic SHOULD be implemented as ordinary Python functions or services. + +Framework-specific wrappers SHALL remain thin. + +The conceptual structure is: + +```text +Reusable operation + ├── called synchronously where CARE requires synchronous behavior + ├── called by a Celery task wrapper + ├── called by a Cloud Tasks HTTP handler + └── called by another approved backend wrapper +``` + +A function SHALL not become asynchronous merely because it is decorated as a Celery task today. + +Existing call-site semantics SHALL be preserved unless an explicit implementation specification changes them. + +## Dispatch contract + +Asynchronous producers SHALL use a narrow dispatch contract conceptually equivalent to: + +```python +enqueue_task( + task_name, + payload, + delay_seconds=None, + task_id=None, +) +``` + +The exact API SHALL remain limited to verified CARE requirements. + +The initial contract SHALL NOT attempt to reproduce all Celery features. + +Unless verified call sites require them, the abstraction SHALL not include: + +- chains; +- chords; +- groups; +- canvases; +- arbitrary callbacks; +- arbitrary Python import paths; +- general workflow orchestration. + +## Payloads + +Task payloads SHALL be JSON-serializable. + +Payloads SHOULD contain opaque identifiers rather than complete clinical records. + +Task payloads SHALL NOT contain: + +- Django model instances; +- querysets; +- open file handles; +- provider clients; +- credentials; +- unserializable objects; +- complete sensitive records when database identifiers are sufficient. + +Task handlers SHALL reload required state from PostgreSQL. + +## Transaction timing + +When a task depends on a database change, dispatch SHOULD occur after successful commit using Django's transaction facilities, such as: + +```python +transaction.on_commit(...) +``` + +A task SHALL not observe state that was subsequently rolled back. + +## Celery profile + +Celery SHALL remain supported for: + +- local Docker Compose; +- upstream-compatible development; +- traditional deployments; +- installations that intentionally operate a Celery broker and worker. + +Existing Celery task names and signatures SHOULD remain stable where practical. + +Redis may remain Celery's broker and result backend in that profile. + +## Cloud Tasks profile + +Cloud Tasks SHALL be the initial managed asynchronous backend for GCP. + +Cloud Tasks SHALL invoke a private CARE worker over authenticated HTTP. + +The worker SHALL: + +- require platform IAM authentication; +- accept only registered task names; +- reject arbitrary callables; +- validate payloads; +- execute reusable task behavior; +- return success only after successful execution; +- expose retriable failures correctly; +- avoid logging complete sensitive payloads; +- scale to zero when idle. + +Google Cloud Tasks is an implementation of the asynchronous execution contract, not a dependency of CARE business logic. + +## Other providers + +This ADR does not require immediate implementations for AWS, Azure or other clouds. + +Future backends MAY be added when there is a concrete deployment requirement. + +New backends SHALL implement the same narrow behavior required by CARE rather than expanding the contract speculatively. + +## Results + +The asynchronous transport SHALL not be treated as the durable application result store. + +Meaningful task results and status SHALL be persisted in: + +- existing domain records; +- report records; +- explicit execution-state records; +- another appropriate PostgreSQL model. + +Cloud Tasks HTTP response bodies SHALL not be used as a result backend. + +Celery results MAY remain for compatibility where verified callers still rely on them, but such dependencies SHALL be inventoried and migrated deliberately. + +## Retries + +Every backend may retry work. + +Retry policy SHALL distinguish: + +- transient infrastructure or external-service failures; +- permanent validation or business failures. + +Framework retry configuration and application exception classification SHALL be mapped explicitly. + +Infinite or uncontrolled retries are prohibited. + +## Idempotency + +Task execution SHALL be treated as at-least-once. + +Handlers SHALL tolerate duplicate delivery where required. + +Idempotency SHOULD rely on: + +- existing database state; +- unique constraints; +- conditional updates; +- explicit idempotency keys; +- execution records; +- object existence where appropriate. + +Redis SHALL not be the sole correctness mechanism. + +## Periodic work + +Periodic scheduling is a distinct concern from request-triggered asynchronous dispatch. + +For the GCP profile: + +- Cloud Scheduler SHALL provide periodic triggers; +- Cloud Run Jobs SHOULD execute maintenance and batch commands; +- Cloud Tasks MAY be used for bounded scheduled work when appropriate. + +For the local Celery profile: + +- Celery Beat MAY remain supported. + +The same production operation SHALL not be scheduled simultaneously by Celery Beat and Cloud Scheduler. + +## Initialization and migrations + +Database migrations, permission synchronization and value-set synchronization SHALL not depend on a permanently running Celery Beat process in the target runtime. + +They SHALL execute through explicit deployment or job commands. + +The API and task worker SHALL not run migrations automatically during normal instance startup. + +## Consequences + +### Positive + +- CARE task behavior is no longer defined by Celery. +- Local Celery compatibility is preserved. +- GCP can use scale-to-zero task execution. +- Only truly asynchronous call sites need transport migration. +- Synchronous calls remain synchronous. +- Task behavior becomes easier to test. +- Future execution backends remain possible. + +### Negative + +- Thin wrappers must be maintained for multiple enabled backends. +- Retry semantics differ between implementations. +- Task results must move to explicit application state. +- Handlers require idempotency review. +- Periodic scheduling must be managed separately. +- Not every Celery feature is portable. + +## Alternatives Considered + +### Replace every Celery task with Cloud Tasks + +Rejected. + +Task decorators do not prove asynchronous intent, and several current calls execute inline. + +### Retain Celery and use a managed Redis service in every deployment + +Rejected as the only architecture. + +It requires an active worker and makes Redis mandatory even where managed request-driven execution is preferable. + +### Use Cloud Tasks directly throughout application code + +Rejected. + +It would couple CARE business code to GCP. + +### Build a general workflow engine abstraction + +Rejected. + +CARE's verified requirements do not justify reproducing a workflow platform. + +### Adopt a PostgreSQL queue as the default immediately + +Rejected for the initial GCP profile. + +A PostgreSQL queue requires an active consumer for prompt execution and must be evaluated separately. + +## Out of Scope + +This ADR does not choose: + +- a PostgreSQL task-queue library; +- distributed-lock implementation; +- cache backend; +- Terraform layout; +- detailed Cloud Tasks queue policies; +- workflow orchestration; +- event sourcing; +- cross-cloud task implementations. + +## Related Documents + +- Task call-site inventory +- Cache and Redis inventory +- IS-03: Asynchronous Runtime Modernization +- ADR-0004: Configurable Application Cache +- ADR-0005: Distributed Locking +- ADR-0006: Portable Runtime Profiles + +## Implementation Status + +- [x] Decision accepted. +- [ ] Reusable task logic extracted. +- [ ] Narrow dispatcher implemented. +- [ ] Celery backend preserved. +- [ ] Cloud Tasks backend implemented. +- [ ] Private worker implemented. +- [ ] Periodic work moved to explicit scheduler and jobs. +- [ ] Initialization removed from Celery startup dependency. diff --git a/docs/xii/adr/ADR-0004-configurable-application-cache.md b/docs/xii/adr/ADR-0004-configurable-application-cache.md new file mode 100644 index 0000000000..cbbf46dd89 --- /dev/null +++ b/docs/xii/adr/ADR-0004-configurable-application-cache.md @@ -0,0 +1,228 @@ +# ADR-0004: Configurable Application Cache + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +## Context + +CARE currently configures Redis as its default Django cache. + +Repository inspection identified multiple operations that appear related to Redis but do not all represent ordinary cache behavior: + +- standard cache reads and writes; +- report-progress values; +- rate-limit counters; +- `cache.set(..., nx=True)` used as lock-like behavior; +- `cache.delete_pattern(...)`; +- direct `get_redis_connection()` access; +- Celery broker and result storage; +- health checks. + +A backend swap from Redis to PostgreSQL or LocMem cannot safely replace all these responsibilities. + +The existing LocMem shim accepts an `nx` argument while always returning success, silently removing mutual exclusion. This demonstrates that cache configuration and distributed locking must be separated. + +The project wants Redis to be optional, while supporting: + +- PostgreSQL-backed shared cache; +- LocMem for process-local performance values; +- Redis-compatible shared cache where beneficial; +- the existing local Redis profile. + +## Decision + +CARE SHALL use Django's cache framework as the abstraction for disposable cached values. + +Cache backend selection SHALL be configuration-driven. + +Initial supported cache profiles SHALL include: + +```text +postgres +redis +locmem +dummy +``` + +The default local upstream-compatible profile MAY continue using Redis. + +The initial low-service-count cloud profile MAY use Django's PostgreSQL database cache. + +Cache SHALL NOT be used as a generic substitute for: + +- distributed locks; +- durable application state; +- task queues; +- correctness-critical coordination; +- arbitrary Redis commands. + +## Cache semantics + +Values stored through the cache API SHALL be disposable. + +Deleting or losing all cache entries SHALL not destroy durable CARE state. + +Correctness-critical or auditable state SHALL use explicit PostgreSQL models or constraints. + +## PostgreSQL cache + +PostgreSQL cache SHALL use Django's supported database-cache backend. + +It is appropriate for: + +- moderate shared cache traffic; +- cross-instance disposable values; +- regenerated configuration; +- selected progress values where expiration is acceptable; +- avoiding a separate Redis service in smaller deployments. + +It is not assumed to match Redis latency or throughput. + +The cache table SHALL be initialized explicitly during environment setup. + +## Redis cache + +Redis-compatible storage MAY be used for: + +- higher-frequency shared cache; +- lower-latency counters; +- deployments that already operate Redis; +- workloads where PostgreSQL cache pressure becomes excessive. + +Configuration SHALL remain provider-neutral. + +Upstash or another compatible service may be selected through standard Redis URLs. + +## LocMem + +LocMem MAY be used only for values that do not require cross-process or cross-instance consistency. + +It is appropriate for: + +- process-local performance optimization; +- regenerated schema data; +- test or development scenarios. + +It SHALL NOT be used for: + +- distributed locking; +- globally enforced rate limits; +- shared task progress; +- correctness-sensitive state. + +## Report progress + +Report progress SHALL be classified separately. + +It may use: + +- the configured shared cache when disposable progress is sufficient; +- an explicit PostgreSQL model when durability, auditability or failure history is required. + +Progress values SHALL not be described or implemented as locks. + +## Nonportable cache operations + +Operations such as: + +```text +delete_pattern +get_redis_connection +backend-specific command execution +``` + +SHALL not appear in provider-neutral cache consumers. + +Each existing use SHALL be: + +- eliminated; +- replaced with explicit key tracking; +- moved to a responsibility-specific implementation; +- retained only inside a Redis-specific optional component. + +## Failure behavior + +Every cache use SHALL define whether cache failure: + +- becomes a cache miss; +- produces a controlled degraded response; +- blocks the operation. + +Performance caches may fail open as misses. + +Correctness-sensitive behavior must not rely on ignored cache exceptions. + +## Consequences + +### Positive + +- Redis becomes optional for ordinary caching. +- PostgreSQL can provide moderate shared cache without another service. +- Cache consumers align with Django. +- Provider-specific Redis operations are isolated. +- Locks and durable state are no longer confused with caching. + +### Negative + +- PostgreSQL cache adds database queries and table growth. +- Different profiles have different latency characteristics. +- Existing backend-specific cache operations require refactoring. +- Some values may need dedicated models instead of cache. + +## Alternatives Considered + +### Replace Redis globally with DatabaseCache + +Rejected. + +Redis currently performs responsibilities beyond caching. + +### Keep Redis mandatory + +Rejected. + +Smaller cloud-native deployments should not require it for ordinary caching. + +### Use LocMem as the default cloud cache + +Rejected for shared values. + +Cloud Run instances do not share LocMem state. + +### Create a custom generic cache API + +Rejected. + +Django already provides the required abstraction for cache semantics. + +## Out of Scope + +This ADR does not define: + +- distributed locks; +- Celery broker selection; +- task queues; +- detailed rate-limit implementation; +- exact report-progress model; +- database sizing; +- Upstash-specific features. + +## Related Documents + +- Cache and Redis inventory +- ADR-0003: Configurable Asynchronous Execution +- ADR-0005: Distributed Locking +- IS-04: Cache Modernization + +## Implementation Status + +- [x] Decision accepted. +- [ ] Cache responsibilities classified. +- [ ] PostgreSQL cache implemented. +- [ ] Redis cache retained as optional. +- [ ] LocMem use restricted. +- [ ] Backend-specific operations removed from generic consumers. +- [ ] Report progress assigned to an appropriate backend. diff --git a/docs/xii/adr/ADR-0005-distributed-locking.md b/docs/xii/adr/ADR-0005-distributed-locking.md new file mode 100644 index 0000000000..7b16a8d4ac --- /dev/null +++ b/docs/xii/adr/ADR-0005-distributed-locking.md @@ -0,0 +1,182 @@ +# ADR-0005: Distributed Locking as a Separate Responsibility + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +## Context + +CARE currently implements lock-like behavior through cache and Redis-related operations. + +Verified patterns include: + +- `cache.set(..., nx=True)`; +- backend-specific arguments; +- direct Redis access; +- a LocMem compatibility implementation that accepts `nx` but always succeeds. + +This can silently remove mutual exclusion when the cache backend changes. + +A lock is not a cache value. + +The project must support portable deployment profiles without allowing backend substitution to weaken concurrency correctness. + +## Decision + +Distributed locking SHALL be treated as an independent infrastructure responsibility. + +Lock acquisition and release SHALL not depend on undocumented cache-backend extensions. + +CARE SHALL use an explicit locking interface limited to verified application requirements. + +The lock implementation SHALL provide clearly defined semantics for: + +- acquisition; +- contention; +- timeout; +- expiration or lease behavior; +- release; +- ownership; +- failure; +- process termination; +- duplicate release. + +The final initial backend SHALL be selected only after every lock call site is analyzed. + +## Backend candidates + +Acceptable candidates for evaluation include: + +- PostgreSQL advisory locks; +- PostgreSQL row-level locking; +- unique constraints and transactional state transitions; +- Redis locks; +- removal of a lock where a database constraint is the correct mechanism. + +Different call sites MAY require different mechanisms. + +The project SHALL not force every concurrency problem through one generic distributed-lock service. + +## Database correctness first + +When the invariant can be enforced using: + +- unique constraints; +- conditional updates; +- `SELECT ... FOR UPDATE`; +- transaction isolation; +- idempotency records; + +those database mechanisms SHOULD be preferred over a cache lock. + +A lock SHALL not replace a durable database invariant. + +## PostgreSQL advisory locks + +PostgreSQL advisory locks MAY be used when: + +- all contenders share PostgreSQL; +- lock scope can be represented safely; +- session or transaction lifetime is understood; +- connection-pool behavior is compatible; +- failure and cleanup semantics are tested. + +They SHALL not be adopted automatically for every call site. + +## Redis locks + +Redis-compatible locks MAY remain an option when: + +- low-latency distributed coordination is genuinely required; +- Redis is enabled for the deployment profile; +- lease expiration and ownership are implemented correctly; +- provider command support is verified. + +Redis SHALL not be mandatory solely because existing code used `nx`. + +## No fake locking + +A backend SHALL not claim success without providing actual exclusion. + +Unsupported lock behavior SHALL fail explicitly. + +LocMem may provide process-local locks only when the documented scope is explicitly process-local. It SHALL not impersonate a distributed lock. + +## Observability + +Lock contention and timeout SHOULD be observable without logging sensitive records. + +Useful metadata includes: + +- lock category; +- opaque resource identifier; +- wait duration; +- timeout; +- acquisition result; +- process role. + +## Consequences + +### Positive + +- Cache swaps cannot silently remove locking. +- Concurrency semantics become explicit. +- Database invariants can replace unnecessary locks. +- Redis remains optional where not required. +- Lock behavior becomes testable. + +### Negative + +- Every lock call site requires analysis. +- More than one coordination mechanism may remain. +- PostgreSQL advisory locks require careful connection handling. +- Redis locks require correct lease and ownership behavior. + +## Alternatives Considered + +### Keep `cache.set(nx=True)` + +Rejected. + +It is not part of Django's portable cache contract and already fails silently under the current LocMem shim. + +### Use one universal Redis lock + +Rejected as the default. + +It would make Redis mandatory and could hide database-integrity problems. + +### Use one universal PostgreSQL advisory-lock service + +Rejected before call-site analysis. + +Some invariants are better enforced with constraints or row locks. + +### Remove all locks + +Rejected. + +Some operations may genuinely require mutual exclusion. + +## Out of Scope + +This ADR does not select the final backend for every lock. + +That selection belongs to IS-05 after call-site analysis and concurrency tests. + +## Related Documents + +- Cache and Redis inventory +- ADR-0004: Configurable Application Cache +- IS-05: Distributed Lock Modernization + +## Implementation Status + +- [x] Decision accepted. +- [ ] Lock call sites classified. +- [ ] LocMem false-lock behavior removed. +- [ ] Database invariants identified. +- [ ] Approved lock mechanisms implemented. +- [ ] Concurrency tests completed. diff --git a/docs/xii/adr/ADR-0006-portable-runtime-profiles.md b/docs/xii/adr/ADR-0006-portable-runtime-profiles.md new file mode 100644 index 0000000000..c31ee43cca --- /dev/null +++ b/docs/xii/adr/ADR-0006-portable-runtime-profiles.md @@ -0,0 +1,215 @@ +# ADR-0006: Portable Runtime Profiles with GCP as the First Managed Target + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +## Context + +The project began with the practical goal of running CARE inexpensively on GCP without a permanently running VM. + +During design, the broader requirement became explicit: + +- CARE must remain usable locally; +- MinIO, Redis and Celery must remain supported for local or traditional deployments; +- application code should depend on Django and narrow internal contracts rather than a cloud provider; +- GCP is the first managed-cloud profile, not the only possible deployment. + +The current local runtime uses continuously running Docker Compose services. + +The initial GCP target should use managed or request-driven services where practical. + +## Decision + +CARE SHALL support explicit runtime profiles. + +The initial profiles are: + +### Local or traditional profile + +```text +Django application +PostgreSQL +MinIO through Django Storage +Redis +Celery +Celery Beat +``` + +### Initial GCP profile + +```text +Cloud Run API +Cloud SQL for PostgreSQL +Cloud Storage through Django Storage +Cloud Tasks +private Cloud Run task worker +Cloud Scheduler +Cloud Run Jobs +Secret Manager +Artifact Registry +Cloud Logging +optional Redis-compatible services +``` + +Application behavior SHALL remain provider-neutral. + +Cloud-specific integrations SHALL be isolated in: + +- deployment settings; +- backend implementations; +- startup commands; +- infrastructure code. + +## Django and PostgreSQL + +Django and Django ORM remain intentional architecture choices. + +PostgreSQL remains CARE's durable system of record. + +No repository abstraction over Django ORM is required. + +## Runtime roles + +The managed-cloud application image SHOULD support distinct roles: + +```text +API +HTTP task worker +management or batch job +optional Celery worker +optional PostgreSQL queue worker +``` + +The same immutable image SHOULD be reused where practical. + +Roles differ through commands, configuration, IAM and scaling. + +## Initialization + +Migrations and setup commands SHALL run explicitly through deployment jobs. + +Normal API or worker instance startup SHALL not: + +- run migrations; +- synchronize permissions; +- synchronize value sets; +- depend on Celery Beat. + +## Scale to zero + +The initial GCP profile SHOULD allow these components to scale to zero: + +- API, when operational requirements permit; +- Cloud Tasks worker; +- Cloud Run Jobs when not executing. + +Cloud SQL does not scale to zero and represents a baseline cost. + +The architecture SHALL not claim the entire system is serverless or zero-cost when idle. + +## Portability + +Runtime-specific choices SHALL not leak into CARE domain logic. + +GCP support SHALL not require: + +- GCP credentials locally; +- GCS locally; +- Cloud Tasks locally; +- Cloud Run-specific branching in business code. + +Future runtime profiles may be added through explicit implementation decisions. + +## Redis + +Redis SHALL remain optional in the GCP profile. + +It may be enabled for selected responsibilities such as high-frequency cache or coordination after the appropriate ADRs and specifications are implemented. + +## Security + +The GCP profile SHALL use: + +- private storage buckets; +- least-privilege service accounts; +- private worker invocation; +- Secret Manager; +- controlled Cloud SQL connectivity; +- no committed service-account keys; +- no direct frontend storage credentials. + +## Consequences + +### Positive + +- Local and traditional operation remain supported. +- GCP can avoid permanent VMs and polling workers. +- Application logic remains portable. +- Cloud resources can use managed identities. +- Deployment roles become explicit. + +### Negative + +- Multiple supported profiles increase testing requirements. +- Configuration validation becomes more complex. +- Runtime capabilities differ between profiles. +- Cloud SQL remains a persistent cost. +- Operational documentation must distinguish profiles. + +## Alternatives Considered + +### Make GCP the only supported runtime + +Rejected. + +The application must remain locally usable and portable. + +### Preserve the current Docker Compose topology in a VM + +Rejected for the initial GCP profile. + +It creates a continuously running VM and higher operational burden. + +### Use Kubernetes as the universal runtime + +Rejected. + +It introduces unnecessary complexity for the initial requirements. + +### Abstract every cloud service immediately + +Rejected. + +Only concrete supported profiles should be implemented. + +## Out of Scope + +This ADR does not define: + +- detailed Terraform modules; +- CI/CD; +- exact Cloud Run sizing; +- exact Cloud SQL tier; +- AWS or Azure profiles; +- multi-region architecture. + +## Related Documents + +- ADR-0001 through ADR-0005 +- IS-06: Runtime Profiles +- Operations guide +- Configuration reference + +## Implementation Status + +- [x] Decision accepted. +- [ ] Production container implemented. +- [ ] GCP settings implemented. +- [ ] Cloud SQL integrated. +- [ ] API deployed to Cloud Run. +- [ ] Private worker deployed. +- [ ] Jobs and Scheduler implemented. +- [ ] Redis-free GCP profile verified. diff --git a/docs/xii/adr/ADR-0007-terraform-for-GCP.md b/docs/xii/adr/ADR-0007-terraform-for-GCP.md new file mode 100644 index 0000000000..284cd11cdc --- /dev/null +++ b/docs/xii/adr/ADR-0007-terraform-for-GCP.md @@ -0,0 +1,201 @@ +# ADR-0007: Terraform for GCP Infrastructure as Code + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +## Context + +The initial managed-cloud profile requires multiple coordinated GCP resources: + +- APIs; +- IAM identities; +- Cloud Run services; +- Cloud Run Jobs; +- Cloud SQL; +- Cloud Storage; +- Cloud Tasks; +- Cloud Scheduler; +- Secret Manager; +- Artifact Registry; +- networking; +- monitoring. + +Manual creation would make environments difficult to reproduce, audit, review and destroy safely. + +The deployment is greenfield and can be created directly from declared infrastructure. + +## Decision + +Terraform SHALL be the infrastructure-as-code tool for the initial GCP profile. + +Terraform SHALL manage the lifecycle and configuration of supported GCP infrastructure. + +Application source code SHALL not create production infrastructure at runtime. + +## Scope + +Terraform SHALL manage, as applicable: + +- required GCP APIs; +- Artifact Registry; +- service accounts; +- IAM bindings; +- Cloud SQL; +- databases or database users where appropriate; +- Cloud Storage buckets; +- Cloud Tasks queues; +- Cloud Scheduler jobs; +- Cloud Run API; +- Cloud Run worker; +- Cloud Run Jobs; +- Secret Manager resources and access bindings; +- required networking; +- monitoring and alerting foundations; +- environment-specific resource configuration. + +## Application schemas + +Terraform SHALL not contain embedded application SQL for creating: + +- Django tables; +- database-cache tables; +- PostgreSQL queue schemas; +- application models. + +Those SHALL be created through: + +- Django migrations; +- Django management commands; +- approved queue-library schema tooling. + +## Environments + +Infrastructure SHALL support separate environment compositions for: + +```text +dev +staging +prod +``` + +Environment separation SHALL include stateful resources and secrets. + +Module reuse is encouraged, but modules SHALL not be introduced solely for aesthetic abstraction. + +## State + +Terraform state SHALL be remote and protected. + +State storage SHALL: + +- restrict access; +- support recovery or versioning; +- separate environments; +- be treated as sensitive infrastructure metadata. + +## Plans and review + +Production applies SHOULD use reviewed Terraform plans. + +Plans SHALL be inspected for destructive changes, especially: + +- Cloud SQL replacement; +- bucket deletion; +- secret deletion; +- IAM broadening; +- service-account replacement; +- queue deletion; +- networking changes. + +## Resource protection + +Stateful production resources SHOULD use appropriate protections, such as: + +- deletion protection; +- lifecycle restrictions; +- bucket retention decisions; +- backup configuration. + +Terraform destroy is not an application rollback mechanism. + +## Secrets + +Terraform may create secret resources and IAM bindings. + +Secret values SHOULD be injected through a secure operational process and SHALL not be committed as plaintext Terraform variables. + +Sensitive outputs SHALL be minimized. + +## Consequences + +### Positive + +- Environments are reproducible. +- Infrastructure changes are reviewable. +- IAM and networking are versioned. +- Greenfield environments can be recreated. +- Drift becomes easier to detect. +- Deployment documentation can reference concrete code. + +### Negative + +- Terraform state must be protected. +- Provider and module versions require maintenance. +- Stateful-resource changes require careful planning. +- Some operational secret workflows remain outside Terraform. + +## Alternatives Considered + +### Manual GCP configuration + +Rejected. + +It is not reproducible or sufficiently auditable. + +### Pulumi + +Not selected. + +Terraform has broad GCP support and matches the current project plan. + +### Kubernetes manifests + +Rejected. + +Kubernetes is not the selected runtime. + +### Application-created infrastructure + +Rejected. + +Infrastructure lifecycle must remain outside application execution. + +## Out of Scope + +This ADR does not define: + +- the exact Terraform directory implementation; +- CI deployment permissions; +- an organization-wide GCP landing zone; +- multi-cloud infrastructure code; +- application database migrations. + +## Related Documents + +- ADR-0006: Portable Runtime Profiles +- IS-07: Infrastructure as Code +- GCP Operations Guide +- Configuration Reference + +## Implementation Status + +- [x] Decision accepted. +- [ ] Remote state configured. +- [ ] Environment layout implemented. +- [ ] GCP resources declared. +- [ ] IAM validated. +- [ ] Development environment applied. +- [ ] Destructive-change protections tested. diff --git a/docs/xii/adr/ADR-0008-automated-continuous-integration.md b/docs/xii/adr/ADR-0008-automated-continuous-integration.md new file mode 100644 index 0000000000..283a4070ce --- /dev/null +++ b/docs/xii/adr/ADR-0008-automated-continuous-integration.md @@ -0,0 +1,218 @@ +# ADR-0008: Automated Continuous Integration and Controlled Delivery + +- **Status:** Accepted +- **Date:** 2026-08-06 +- **Decision Makers:** CARE Fork Maintainers +- **Supersedes:** None +- **Superseded by:** None + +## Context + +The maintained CARE fork must: + +- preserve the upstream-compatible local runtime; +- test multiple runtime profiles; +- produce immutable container images; +- deploy explicit database migrations; +- deploy worker and API revisions in a compatible order; +- validate Terraform; +- record the upstream base; +- support controlled rollback. + +Manual build and deployment would make releases inconsistent and difficult to audit. + +The repository is hosted on GitHub and can use GitHub Actions unless a later operational constraint requires another CI system. + +## Decision + +The project SHALL implement automated continuous integration and controlled delivery. + +GitHub Actions SHALL be the initial CI/CD platform. + +CI SHALL validate every relevant change. + +Production delivery SHALL remain gated and auditable. + +## Continuous integration + +CI SHALL include appropriate stages for: + +- formatting; +- linting; +- upstream CARE tests; +- GCP or portable-settings tests; +- storage tests; +- file-transport tests; +- task tests; +- cache and lock tests as implemented; +- container build; +- Terraform formatting and validation; +- security or secret scanning where practical. + +Not every live-cloud integration test must run on untrusted pull requests. + +Credentialed tests may run on: + +- protected branches; +- approved workflows; +- staging deployment; +- scheduled tests. + +## Immutable images + +The pipeline SHALL build an immutable container image. + +Images SHALL be identified using: + +- Git commit SHA; +- release tag; +- image digest. + +The API, task worker and jobs SHOULD use the same image revision. + +## Deployment order + +The controlled deployment sequence SHALL be: + +1. validate code and infrastructure; +2. build and publish the image; +3. update the migration job; +4. run migrations; +5. stop on migration failure; +6. deploy the task worker; +7. deploy the API; +8. update jobs and schedules; +9. run smoke tests; +10. record release metadata. + +The worker SHALL normally deploy before an API revision that produces new task payloads. + +## Migrations + +Migrations SHALL be explicit deployment operations. + +API or worker startup SHALL not be the migration mechanism. + +The pipeline SHALL not deploy the new API when migrations fail. + +Destructive migrations SHOULD use expand-and-contract techniques where practical after real production data exists. + +## Environment promotion + +Changes SHOULD be verified in staging before production. + +Production deployment SHOULD require an approved workflow or environment gate. + +The exact approval model may depend on the operating organization. + +## Secrets and identity + +CI/CD SHALL use protected deployment identity. + +Long-lived static service-account keys SHOULD be avoided. + +Workload identity federation or another short-lived identity mechanism SHOULD be preferred where supported. + +Secrets SHALL not be printed in logs. + +## Rollback + +Application rollback SHALL deploy a previous immutable revision. + +The pipeline SHALL record enough metadata to identify: + +- application commit; +- upstream commit; +- image digest; +- Terraform commit; +- migration state. + +Database rollback is separate and SHALL not happen automatically. + +## Upstream synchronization + +CI SHALL validate synchronization branches against: + +- the local upstream-compatible profile; +- the selected managed-cloud profile; +- container build; +- Terraform validation. + +The release metadata SHALL record the upstream base commit. + +## Consequences + +### Positive + +- Releases become repeatable. +- Tests gate changes. +- Images are traceable. +- Migration failures stop deployment. +- Worker/API compatibility can be controlled. +- Rollback references are preserved. +- Upstream synchronization is safer. + +### Negative + +- CI workflows require maintenance. +- Live integration tests require secure credentials. +- Multiple profiles increase build time. +- Database migrations still require human judgment. +- Pipeline permissions are security-sensitive. + +## Alternatives Considered + +### Manual deployment + +Rejected. + +It is error-prone and difficult to audit. + +### Automatically deploy every branch to production + +Rejected. + +Production requires controlled promotion. + +### Run migrations during container startup + +Rejected. + +Multiple concurrent instances and failed startup can create unsafe deployment behavior. + +### Build separate images for every role + +Rejected as the default. + +A shared immutable image reduces drift unless role-specific needs later justify separation. + +## Out of Scope + +This ADR does not define: + +- the exact YAML workflows; +- organization-wide release governance; +- a particular branching service beyond the documented fork strategy; +- automated database rollback; +- multi-cloud delivery pipelines. + +## Related Documents + +- ADR-0006: Portable Runtime Profiles +- ADR-0007: Terraform for GCP Infrastructure as Code +- IS-08: Continuous Delivery +- Testing Strategy +- Upstream Synchronization +- Operations Guide + +## Implementation Status + +- [x] Decision accepted. +- [ ] CI checks implemented. +- [ ] Immutable image build implemented. +- [ ] Terraform validation implemented. +- [ ] Staging deployment implemented. +- [ ] Migration gate implemented. +- [ ] Worker/API deployment order implemented. +- [ ] Production approval implemented. +- [ ] Release metadata recorded. diff --git a/docs/xii/architecture/00-scope-and-goals.md b/docs/xii/architecture/00-scope-and-goals.md new file mode 100644 index 0000000000..e9dc850349 --- /dev/null +++ b/docs/xii/architecture/00-scope-and-goals.md @@ -0,0 +1,882 @@ +--- +title: Scope and Goals +document: 00-scope-and-goals +version: 0.1.0 +status: Draft +authors: + - César Benjamín García Martínez +--- + +# CARE GCP Deployment and Compatibility Guide + +## 1. Purpose + +This guide defines how to deploy and operate CARE on Google Cloud Platform while preserving compatibility with the official upstream repository. + +The objective is not to redesign CARE. + +The objective is to adapt its infrastructure so that it can run with low operational overhead, avoid permanently running virtual machines where practical, and continue receiving future upstream updates with minimal conflict. + +This guide is intentionally limited to deployment, infrastructure integration and the smallest application changes required to support that deployment model. + +--- + +## 2. Primary Goal + +The primary goal is to run CARE using managed or serverless Google Cloud services. + +The target production architecture is: + +- CARE backend on Cloud Run +- PostgreSQL on Cloud SQL +- clinical and facility files on Cloud Storage +- asynchronous tasks on Cloud Tasks +- asynchronous task execution on a private Cloud Run worker +- periodic tasks on Cloud Scheduler +- batch and maintenance operations on Cloud Run Jobs +- container images on Artifact Registry +- secrets on Secret Manager +- logs on Cloud Logging + +The architecture SHOULD avoid: + +- permanently running Compute Engine virtual machines +- self-managed PostgreSQL +- self-managed MinIO +- self-managed Redis +- permanently running Celery workers + +--- + +## 3. Core Constraint + +The fork MUST remain maintainable against: + +```text +https://github.com/ohcnetwork/care +```` + +The official `develop` branch is treated as upstream. + +Changes introduced by this project MUST minimize divergence from upstream. + +Whenever possible, the implementation SHOULD add new files instead of heavily modifying existing upstream files. + +--- + +## 4. What This Project Is + +This project is an infrastructure adaptation of CARE. + +It adds support for: + +* Google Cloud deployment +* Google Cloud Storage +* Cloud Tasks +* Cloud Run +* Cloud SQL +* Cloud Scheduler +* Cloud Run Jobs +* optional Redis-compatible services +* Terraform or equivalent infrastructure as code +* deployment automation +* upstream synchronization procedures + +It MAY add small internal adapters where CARE currently depends directly on infrastructure that must be replaced in GCP. + +--- + +## 5. What This Project Is Not + +This project is not: + +* a rewrite of CARE +* a redesign of the clinical domain +* a replacement for Django +* a replacement for Django ORM +* a migration away from PostgreSQL +* a full Clean Architecture conversion +* a Domain-Driven Design restructuring +* a repository-pattern migration +* a provider-neutral framework for every possible cloud +* a general-purpose infrastructure abstraction library + +The project MUST NOT introduce abstractions unrelated to the deployment objective. + +--- + +## 6. Django and Django ORM + +Django is an accepted and intentional part of the architecture. + +Django ORM remains the standard persistence mechanism. + +Existing code MAY continue to use: + +```python +Patient.objects.get(...) +Encounter.objects.filter(...) +Facility.objects.select_related(...) +``` + +The project MUST NOT introduce repositories for domain models solely to isolate Django ORM. + +Custom managers, querysets or service functions MAY be added when justified by: + +* query reuse +* readability +* performance +* transactional behavior +* existing CARE conventions + +They MUST NOT be added merely for architectural purity. + +--- + +## 7. Upstream Code Ownership + +Existing CARE applications remain owned conceptually by upstream. + +Examples include: + +```text +care/emr/ +care/facility/ +care/users/ +care/security/ +config/ +docker/ +scripts/ +``` + +These files MAY be modified when necessary, but changes SHOULD be: + +* small +* focused +* backwards compatible +* covered by tests +* easy to reapply after upstream changes + +Large file moves and package reorganizations are prohibited unless upstream itself performs them. + +--- + +## 8. Deployment Models + +The project supports at least two deployment models. + +### 8.1 Local or traditional deployment + +The existing upstream-compatible stack remains available: + +* Docker Compose +* PostgreSQL +* Redis +* Celery +* MinIO + +This environment is used for: + +* local development +* upstream compatibility tests +* contributor onboarding +* deployments that prefer traditional infrastructure + +### 8.2 GCP deployment + +The GCP deployment uses: + +* Cloud Run +* Cloud SQL +* Cloud Storage +* Cloud Tasks +* Cloud Scheduler +* Cloud Run Jobs +* Secret Manager +* Artifact Registry + +Redis MAY be enabled as an optional service for selected responsibilities. + +--- + +## 9. Target Architecture + +```mermaid +flowchart TD + USER[Users and Frontend] --> API[CARE API on Cloud Run] + + API --> SQL[(Cloud SQL PostgreSQL)] + API --> GCS[(Cloud Storage)] + API --> TASKS[Cloud Tasks] + + TASKS --> WORKER[Private CARE Worker on Cloud Run] + WORKER --> SQL + WORKER --> GCS + + SCHEDULER[Cloud Scheduler] --> TASKS + SCHEDULER --> JOBS[Cloud Run Jobs] + + JOBS --> SQL + JOBS --> GCS + + SECRETS[Secret Manager] --> API + SECRETS --> WORKER + SECRETS --> JOBS + + REGISTRY[Artifact Registry] --> API + REGISTRY --> WORKER + REGISTRY --> JOBS + + REDIS[(Optional Redis-compatible service)] + API -. optional cache, rate limits or transient state .-> REDIS + WORKER -. optional locks or transient state .-> REDIS +``` + +--- + +## 10. Service Replacement Map + +The initial replacement strategy is: + +| Existing local component | GCP production component | +| ------------------------ | ------------------------------------------------ | +| Django backend container | Cloud Run service | +| PostgreSQL container | Cloud SQL for PostgreSQL | +| MinIO | Cloud Storage | +| Redis as Celery broker | Cloud Tasks | +| Celery worker | Private Cloud Run worker | +| Celery Beat | Cloud Scheduler | +| maintenance commands | Cloud Run Jobs | +| local secrets or `.env` | Secret Manager and runtime environment variables | +| locally built images | Artifact Registry | + +Redis is not replaced globally. + +Only its use as the Celery broker is replaced by Cloud Tasks in the default GCP deployment. + +--- + +## 11. Redis Policy + +Redis becomes optional in GCP. + +Redis MUST NOT remain a mandatory dependency merely because several unrelated CARE features currently use it. + +Its responsibilities MUST be evaluated separately. + +Potential responsibilities include: + +* Celery broker +* Celery result backend +* Django cache +* distributed locks +* rate limiting +* temporary state +* sessions +* direct Redis operations + +The default GCP strategy is: + +| Responsibility | Default GCP backend | Optional alternative | +| ------------------------- | ------------------------------------- | ------------------------ | +| task broker | Cloud Tasks | Celery with Redis | +| task result state | PostgreSQL or domain records | Redis | +| non-critical cache | LocMem | Redis-compatible service | +| distributed locks | *undecided* — see below | Redis-compatible service | +| application rate limiting | PostgreSQL or existing CARE mechanism | Redis-compatible service | +| temporary shared state | PostgreSQL | Redis-compatible service | +| sessions | existing Django configuration | Redis | + +Distributed locking has no default yet. PostgreSQL advisory locks are the +leading candidate, but they are session-scoped rather than TTL-scoped, which is +a different failure model from the Redis locks CARE holds today: a lock is +released when the connection drops instead of when a timeout expires, and a +pooler that multiplexes connections can hand the same session to another +request. Naming a default before each call site has been examined against that +difference would be guessing. + +A default SHALL be recorded here only once an ADR covering distributed locking +carries, for every existing lock: what it protects, whether correctness or +merely duplicated work is at stake, and evidence of its concurrency and hold +time under load. + +--- + +## 12. Redis-Compatible Providers + +When Redis is enabled, the implementation SHOULD use standard Redis protocols and avoid depending on one vendor. + +Potential providers include: + +* Upstash Redis +* Google Memorystore +* Redis OSS +* Valkey +* Dragonfly +* KeyDB + +The application SHOULD use provider-neutral configuration such as: + +```env +CARE_CACHE_BACKEND=redis +REDIS_CACHE_URL=rediss://... +``` + +It SHOULD NOT use provider-specific names such as: + +```env +USE_UPSTASH=true +``` + +unless functionality is genuinely unique to that provider. + +--- + +## 13. Upstash + +Upstash MAY be used for responsibilities compatible with its service model. + +Reasonable uses include: + +* shared cache +* rate limiting counters +* non-critical transient state +* short-lived coordination + +Upstash SHOULD NOT automatically become: + +* the Cloud Tasks replacement +* a reason to retain permanently running Celery workers +* the primary durable store +* the only integrity mechanism for clinical operations + +Sensitive data stored in any external Redis-compatible service MUST be minimized. + +Keys and values SHOULD use opaque identifiers rather than patient names, diagnoses, clinical notes or complete task payloads. + +Any production use MUST be reviewed against applicable privacy, contractual and data-residency requirements. + +--- + +## 14. MinIO and Cloud Storage + +MinIO remains supported for local development. + +Production GCP SHOULD use Cloud Storage. + +Django `FileField` and `default_storage` SHOULD continue using standard Django storage APIs whenever possible. + +A custom storage adapter SHOULD only be introduced for operations that are not adequately covered by Django storage, such as: + +* multipart upload coordination +* direct bucket operations +* nonstandard object metadata operations + +The project MUST NOT create an elaborate storage abstraction if changing Django `STORAGES` is sufficient. + +**Superseded by ADR-0001.** Earlier revisions of this list also named +*provider-specific signed upload flows* and *direct `boto3` usage*. Neither is a +permitted reason any more: ADR-0001 removed signed-URL transport outright and +forbids provider SDK use in application code. CARE mediates every transfer, so a +client never receives a storage-provider URL in either direction. + +--- + +## 15. Celery and Cloud Tasks + +Celery remains supported. + +The local stack MAY continue to use: + +```text +Redis + Celery worker + Celery Beat +``` + +The default GCP stack SHOULD use: + +```text +Cloud Tasks + private Cloud Run worker + Cloud Scheduler +``` + +A small task-dispatch adapter MAY be introduced to choose between: + +```env +CARE_TASK_BACKEND=celery +``` + +and: + +```env +CARE_TASK_BACKEND=cloud_tasks +``` + +The adapter MUST remain limited to task dispatch and task execution concerns. + +It MUST NOT become a general application framework. + +--- + +## 16. Cloud Run Services + +The production deployment SHOULD use separate Cloud Run services for: + +### CARE API + +Receives user and frontend requests. + +### CARE task worker + +Receives authenticated task requests from Cloud Tasks. + +The worker SHOULD: + +* deny unauthenticated access +* accept only explicitly registered task handlers +* use OIDC-based invocation +* scale to zero +* return non-2xx responses for retriable failures + +Both services SHOULD use the same container image when practical. + +--- + +## 17. Cloud Run Jobs + +Cloud Run Jobs SHOULD execute operations that do not belong in request-serving services. + +Examples include: + +* database migrations +* fixture loading +* value-set synchronization +* bulk cleanup +* data imports +* scheduled batch processing +* administrative Django commands + +Database migrations MUST NOT run automatically every time a Cloud Run API instance starts. + +--- + +## 18. Cloud SQL + +Cloud SQL remains a continuously provisioned service and generally does not scale to zero. + +This is accepted because PostgreSQL is CARE's durable system of record. + +The deployment SHOULD minimize its cost without compromising: + +* data durability +* backups +* security +* acceptable performance +* recovery requirements + +Development environments MAY use smaller instances and simplified availability settings. + +Production environments MUST define: + +* automated backups +* point-in-time recovery when required +* deletion protection where appropriate +* private or controlled connectivity +* connection limits +* recovery procedures + +--- + +## 19. Cost Objective + +The architecture SHOULD minimize idle compute cost. + +Services expected to scale to zero include: + +* CARE API on Cloud Run, when traffic permits +* CARE worker on Cloud Run +* Cloud Run Jobs +* Cloud Tasks consumers + +Services that may produce persistent cost include: + +* Cloud SQL +* stored Cloud Storage data +* log retention +* optional Redis services +* network egress +* backups + +The guide MUST clearly distinguish between: + +* resources that scale to zero +* resources billed per use +* resources with permanent baseline cost + +--- + +## 20. Minimal-Change Principle + +Every application modification MUST be justified by a deployment requirement. + +Valid reasons include: + +* replacing MinIO with GCS +* replacing Redis/Celery task dispatch with Cloud Tasks +* making Redis optional +* supporting private task execution +* making health checks compatible with optional services +* supporting Cloud Run startup and proxy behavior + +Invalid reasons include: + +* imposing a new domain architecture +* replacing Django ORM +* moving models into new layers +* renaming existing CARE applications +* introducing abstractions with no current deployment use +* preparing speculative support for unrelated technologies + +--- + +## 21. Abstraction Threshold + +An abstraction SHOULD be introduced only when at least one of these conditions is true: + +1. Two implementations must coexist. + + Example: + + ```text + Celery and Cloud Tasks + ``` + +2. Provider-specific code currently leaks into multiple application modules. + +3. A direct dependency prevents the desired GCP deployment. + +4. Contract tests can meaningfully verify equivalent behavior. + +5. The abstraction reduces upstream modifications. + +An abstraction SHOULD NOT be introduced solely because it appears architecturally elegant. + +--- + +## 22. Branch Strategy + +The recommended branch model is: + +```text +upstream/develop + | + v +origin/develop + | + v +origin/gcp + | + v +feature/* +``` + +### `origin/develop` + +MUST remain an upstream mirror. + +It SHOULD contain no project-specific commits. + +### `origin/gcp` + +Contains the maintained GCP integration. + +It SHOULD remain deployable. + +### `feature/*` + +Contains individual implementation changes. + +### `sync/upstream-YYYY-MM-DD` + +Used to test and resolve upstream merges before merging into `gcp`. + +--- + +## 23. Upstream Synchronization + +The standard synchronization process is: + +```bash +git fetch upstream + +git switch develop +git reset --hard upstream/develop +git push --force-with-lease origin develop + +git switch gcp +git switch -c sync/upstream-YYYY-MM-DD +git merge develop +``` + +After resolving conflicts: + +```bash +make build +make up +make test +``` + +The GCP-specific tests MUST also run before merging the synchronization branch into `gcp`. + +--- + +## 24. Local Compatibility Requirement + +The existing local workflow MUST continue to work. + +At minimum: + +```bash +make build +make up +make load-fixtures +make test +``` + +or the current upstream equivalents. + +GCP-specific changes MUST NOT require cloud credentials for ordinary local development. + +--- + +## 25. Security Scope + +CARE handles sensitive healthcare information. + +The GCP deployment MUST follow these minimum rules: + +* Cloud Storage buckets MUST NOT be public. +* Cloud SQL MUST NOT use unrestricted public access. +* Cloud Run worker MUST NOT allow unauthenticated invocation. +* Service accounts MUST use least privilege. +* Secrets MUST NOT be committed to Git. +* Static service-account key files SHOULD NOT be used in Cloud Run. +* Application logs MUST NOT contain complete clinical payloads. +* Task payloads SHOULD contain identifiers rather than full clinical records. +* External Redis-compatible services MUST NOT receive unnecessary clinical information. +* Object reads MUST be mediated by CARE, which authorizes each request and streams the bytes through Django Storage. CARE issues no signed URL, so there is no expiry window to get wrong and no URL that outlives the permission that produced it (ADR-0001). +* Production access MUST be auditable. + +--- + +## 26. Observability Scope + +The initial implementation SHOULD use native stdout and stderr logging compatible with Cloud Logging. + +It SHOULD include: + +* structured request logs +* task execution logs +* task identifiers +* duration +* failure reason +* retry information +* deployment revision +* environment name + +It MUST avoid logging: + +* passwords +* access tokens +* signed URLs +* complete patient records +* clinical files +* secret values + +Advanced tracing and metrics MAY be added later. + +They are not required for the first deployment. + +--- + +## 27. Infrastructure as Code + +GCP resources SHOULD be managed using Terraform. + +The Terraform configuration SHOULD include: + +* required APIs +* Artifact Registry +* Cloud Run API service +* Cloud Run worker service +* Cloud Run Jobs +* Cloud Tasks queues +* Cloud Scheduler jobs +* Cloud Storage buckets +* Cloud SQL +* Secret Manager +* service accounts +* IAM bindings +* basic monitoring +* environment-specific variables + +Infrastructure code MUST remain separate from CARE business logic. + +--- + +## 28. Environment Separation + +The deployment SHOULD support: + +```text +dev +staging +prod +``` + +Each environment SHOULD have separate: + +* Cloud Run services +* Cloud SQL database or instance +* Cloud Storage buckets +* Cloud Tasks queues +* secrets +* service accounts when appropriate + +Production clinical data MUST NOT be copied into development environments without an approved anonymization process. + +--- + +## 29. Testing Goals + +The project MUST test both deployment modes. + +### Local compatibility tests + +Verify: + +* Docker Compose starts +* Redis and Celery work +* MinIO works +* existing upstream tests pass + +### GCP configuration tests + +Verify: + +* GCP settings load without mandatory Redis +* Cloud Storage configuration is valid +* task backend selection works +* private worker endpoints reject invalid requests +* optional Redis settings validate correctly +* Cloud Run startup commands work +* migrations run as a job + +### Contract tests + +Contract tests SHOULD be limited to real interchangeable components: + +* Celery versus Cloud Tasks dispatch +* MinIO/S3 versus GCS operations where custom adapters exist +* PostgreSQL versus Redis locks if both are implemented + +--- + +## 30. Initial Implementation Order + +The implementation SHOULD proceed in this order: + +1. Inspect the current CARE repository. +2. Document current infrastructure dependencies. +3. Add isolated GCP settings. +4. Deploy the CARE API on Cloud Run with Cloud SQL. +5. Move production media storage to Cloud Storage. +6. Inventory all Celery tasks and direct Redis usage. +7. Add a minimal task-dispatch abstraction. +8. Add the Celery adapter. +9. Add the Cloud Tasks adapter. +10. Add a private Cloud Run worker endpoint. +11. Migrate one low-risk task as a pilot. +12. Migrate remaining compatible tasks. +13. Move periodic tasks to Cloud Scheduler. +14. Move maintenance and batch commands to Cloud Run Jobs. +15. Classify remaining Redis uses. +16. Make Redis optional in GCP settings. +17. Add optional Redis-compatible configuration, including Upstash compatibility. +18. Add Terraform. +19. Add deployment automation. +20. Document upstream synchronization and rollback. + +--- + +## 31. Definition of Done + +The initial GCP adaptation is complete when: + +* CARE runs on Cloud Run. +* Cloud SQL stores application data. +* Cloud Storage stores production files. +* Cloud Tasks handles asynchronous production tasks. +* The Cloud Run worker scales to zero. +* Cloud Scheduler replaces Celery Beat in GCP. +* Cloud Run Jobs handle migrations and batch operations. +* Redis is not required for the default GCP deployment. +* Redis remains supported as an optional backend. +* Upstash or another Redis-compatible service can be configured where appropriate. +* Docker Compose remains functional. +* Celery, Redis and MinIO remain functional locally. +* no Compute Engine VM is required. +* upstream synchronization is documented and tested. +* GCP-specific changes remain small and reviewable. + +--- + +## 32. Out of Scope for the Initial Version + +The following are explicitly outside the initial scope: + +* replacing Django ORM +* introducing repositories for domain models +* moving existing CARE applications +* replacing Django REST Framework +* replacing authentication architecture +* introducing Kubernetes +* supporting every cloud provider +* introducing Temporal +* replacing all existing caches +* redesigning CARE's clinical workflows +* implementing a generic plugin framework +* rewriting all Celery tasks before a pilot succeeds +* creating abstractions for hypothetical future requirements + +--- + +## 33. Governing Rule + +When deciding whether to add a new architectural element, ask: + +> Is this necessary to deploy CARE cheaply and safely on GCP while keeping upstream updates manageable? + +If the answer is no, it does not belong in the initial project. + +--- + +## 34. Next Document + +The next document is: + +```text +docs/xii/architecture/01-current-runtime.md +``` + +It will document the current CARE runtime as it actually exists, including: + +* Docker Compose services +* Django settings +* PostgreSQL +* Redis responsibilities +* Celery configuration +* Celery Beat +* MinIO and S3-compatible storage +* startup scripts +* health checks +* deployment assumptions +* points that must change for Cloud Run diff --git a/docs/xii/architecture/01-current-runtime.md b/docs/xii/architecture/01-current-runtime.md new file mode 100644 index 0000000000..934e7ddebc --- /dev/null +++ b/docs/xii/architecture/01-current-runtime.md @@ -0,0 +1,2001 @@ +--- +title: Current CARE Runtime +document: 01-current-runtime +version: 0.2.1 +status: Draft +source_repository: https://github.com/ohcnetwork/care +source_branch: develop +reviewed: 2026-08-05 +verified_against_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +verified: 2026-08-05 +--- + +# Current CARE Runtime + +## 1. Purpose + +This document describes the current runtime architecture of CARE as implemented +in the official `ohcnetwork/care` repository on the `develop` branch. + +It is a descriptive inventory of the existing system. + +It does not define the target GCP architecture. + +It does not prescribe replacements or migration steps. + +Architectural decisions and migration requirements are documented separately. + +**This document is a baseline, not a description of this fork.** It is pinned to +the `verified_against_commit` in the front matter and describes upstream +`develop` as it was on that date. Work merged into this fork since then has +already replaced parts of what follows — in particular, IS-01 (ADR-0001) +replaced the S3-compatible file manager and its signed URLs with Django Storage +and CARE-served download routes, so §34 and §35 record the *starting* state, not +the current one. The baseline is deliberately left unrevised: it is what the +migration is measured against, and rewriting it would erase that reference +point. For the state after each increment, read the ADRs and the implementation +specifications under `docs/xii/`. + +--- + +## 2. Repository Baseline + +The reviewed repository is: + +```text +https://github.com/ohcnetwork/care +``` + +The default development branch is: + +```text +develop +``` + +The repository contains the CARE backend. + +The application is based on: + +* Django; +* Django REST Framework; +* PostgreSQL; +* Celery; +* Redis; +* S3-compatible object storage; +* Docker Compose; +* environment-based configuration; +* separate settings modules for local, deployment, staging, production and + testing. + +--- + +## 3. Current Runtime Topology + +The local Docker Compose environment consists of five principal services: + +```text +backend +celery +db +redis +minio +``` + +These services are defined across: + +```text +docker-compose.yaml +docker-compose.local.yaml +``` + +The resulting runtime topology is: + +```mermaid +flowchart TD + CLIENT[API client or frontend] --> BACKEND[Django backend] + + BACKEND --> DB[(PostgreSQL)] + BACKEND --> REDIS[(Redis)] + BACKEND --> MINIO[(MinIO)] + + CELERY[Celery worker] --> DB + CELERY --> REDIS + CELERY --> MINIO + + BEAT[Celery Beat embedded in worker] --> CELERY +``` + +The application backend and Celery worker use the same locally built image. + +PostgreSQL, Redis and MinIO are separate infrastructure containers. + +--- + +## 4. Docker Compose Files + +### 4.1 `docker-compose.yaml` + +The base Compose file defines: + +* the shared Docker network; +* PostgreSQL; +* Redis; +* MinIO; +* persistent volumes. + +The default Docker network is explicitly named: + +```text +care +``` + +### 4.2 `docker-compose.local.yaml` + +The local override defines: + +* the Django backend; +* the Celery worker; +* the development image build; +* source-code mounts; +* local environment files; +* development startup scripts. + +The two Compose files are intended to be used together. + +The repository Makefile combines them when executing local development +commands. + +--- + +## 5. PostgreSQL Service + +The database container uses: + +```text +postgres:17-alpine +``` + +Its environment is loaded from: + +```text +docker/.prebuilt.env +``` + +The default local database configuration includes: + +```text +POSTGRES_USER=postgres +POSTGRES_PASSWORD=postgres +POSTGRES_HOST=db +POSTGRES_DB=care +POSTGRES_PORT=5432 +``` + +The Django database URL is: + +```text +postgres://postgres:postgres@db:5432/care +``` + +### 5.1 Persistence + +PostgreSQL data is stored in the named volume: + +```text +postgres-data +``` + +A backup directory is mounted into: + +```text +/backups +``` + +The host-side path defaults to: + +```text +./care-backups +``` + +and can be changed through: + +```text +BACKUP_DIR +``` + +### 5.2 Port exposure + +The container database port is mapped as: + +```text +host 5433 -> container 5432 +``` + +### 5.3 Health check + +The PostgreSQL health check executes: + +```bash +pg_isready -U "${POSTGRES_USER:-postgres}" +``` + +### 5.4 Restart policy + +The service uses: + +```yaml +restart: unless-stopped +``` + +--- + +## 6. Redis Service + +The Redis container uses: + +```text +redis:8-alpine +``` + +### 6.1 Persistence + +Redis data is stored in: + +```text +redis-data +``` + +### 6.2 Port exposure + +Redis is mapped as: + +```text +host 6380 -> container 6379 +``` + +### 6.3 Health check + +The health check executes: + +```bash +redis-cli ping +``` + +### 6.4 Restart policy + +The service uses: + +```yaml +restart: unless-stopped +``` + +--- + +## 7. MinIO Service + +The object-storage container uses: + +```text +minio/minio:latest +``` + +MinIO provides an S3-compatible API for the current local file-storage +implementation. + +### 7.1 Credentials + +The root credentials use: + +```text +MINIO_ACCESS_KEY +MINIO_SECRET_KEY +``` + +with local defaults: + +```text +minioadmin +minioadmin +``` + +### 7.2 Region compatibility + +The container sets: + +```text +AWS_DEFAULT_REGION=ap-south-1 +``` + +The repository comments indicate that this value is used to preserve +compatibility with existing application behavior. + +### 7.3 Storage persistence + +MinIO data is stored in: + +```text +./care/media/minio:/data +``` + +### 7.4 Initialization scripts + +The service mounts: + +```text +docker/minio/init-script.sh +docker/minio/entrypoint.sh +``` + +The custom entrypoint is: + +```text +/entrypoint.sh +``` + +### 7.5 Port exposure + +MinIO exposes: + +```text +host 9100 -> container 9000 +``` + +for the S3-compatible API, and: + +```text +host 9001 -> container 9001 +``` + +for the web console. + +### 7.6 Health check + +The health check requests: + +```text +http://localhost:9000/minio/health/ready +``` + +--- + +## 8. Application Image + +The local application image is named: + +```text +care_local +``` + +It is built from: + +```text +docker/dev.Dockerfile +``` + +The build context is the repository root. + +The build supports: + +```text +ADDITIONAL_PLUGS +``` + +as a build argument. + +The image is not intended to be pulled from Docker Hub or another public +registry. + +Both the backend and Celery services reuse this image. + +--- + +## 9. Django Backend Service + +The backend service: + +* uses the `care_local` image; +* builds the image when required; +* loads `docker/.local.env`; +* mounts the repository into `/app`; +* runs `scripts/start-dev.sh`; +* exposes the Django development server; +* exposes debugpy; +* restarts unless stopped. + +### 9.1 Source mount + +The project root is mounted as: + +```text +.:/app +``` + +This enables live source-code changes during development. + +### 9.2 Ports + +The backend exposes: + +```text +9000 +``` + +for Django and: + +```text +9876 +``` + +for debugpy. + +### 9.3 Dependencies + +The backend declares dependencies on: + +```text +db +redis +celery +``` + +The Celery dependency uses a health condition. + +--- + +## 10. Backend Startup Script + +The development backend starts through: + +```text +scripts/start-dev.sh +``` + +The script performs the following sequence: + +1. Writes the role `api` to `/tmp/container-role`. +2. Waits for PostgreSQL. +3. Waits for Redis. +4. Runs `collectstatic`. +5. Compiles translation messages. +6. Starts Django. + +The normal server command is: + +```bash +python manage.py runserver_plus \ + 0.0.0.0:9000 \ + --print-sql +``` + +When debugger attachment is enabled, the process is started through: + +```text +debugpy +``` + +The backend startup currently assumes that both PostgreSQL and Redis are +available before Django starts. + +--- + +## 11. Celery Service + +The Celery service: + +* uses `care_local`; +* loads `docker/.local.env`; +* mounts the project into `/app`; +* runs `scripts/celery-dev.sh`; +* depends on PostgreSQL and Redis; +* restarts unless stopped. + +--- + +## 12. Celery Startup Script + +The worker starts through: + +```text +scripts/celery-dev.sh +``` + +The script performs: + +1. Writes the role `celery` to `/tmp/container-role`. +2. Waits for PostgreSQL. +3. Waits for Redis. +4. Runs database migrations. +5. Compiles translations. +6. Runs `sync_permissions_roles`. +7. Runs `sync_valueset`. +8. Starts Celery through `watchmedo`. + +The Celery command is: + +```bash +celery \ + --workdir="$(pwd)" \ + -A config.celery_app \ + worker \ + -B \ + --loglevel=INFO +``` + +The `-B` option embeds Celery Beat in the worker. + +The development Celery container therefore performs: + +* initialization and synchronization; +* schema migration; +* asynchronous task processing; +* periodic-task scheduling. + +--- + +## 13. Makefile Runtime Commands + +The repository Makefile defines the local Compose configuration as: + +```text +docker-compose.yaml +docker-compose.local.yaml +``` + +Important commands include: + +```bash +make build +make up +make down +make teardown +make load-fixtures +make list +make logs +make migrate +make test +``` + +### 13.1 `make build` + +Builds the application image using both Compose files. + +### 13.2 `make up` + +Starts the full stack in detached mode and waits for service readiness. + +### 13.3 `make down` + +Stops and removes containers while preserving volumes. + +### 13.4 `make teardown` + +Stops containers and deletes volumes. + +This removes persisted PostgreSQL and Redis data. + +### 13.5 `make load-fixtures` + +Runs: + +```bash +python manage.py load_fixtures +``` + +inside the backend container. + +### 13.6 Database utilities + +The Makefile also provides commands for: + +* dumping PostgreSQL; +* restoring PostgreSQL; +* resetting the database; +* running migrations; +* checking migrations. + +--- + +## 14. Django Settings Modules + +The repository contains: + +```text +config/settings/__init__.py +config/settings/base.py +config/settings/config.py +config/settings/deployment.py +config/settings/local.py +config/settings/production.py +config/settings/staging.py +config/settings/test.py +``` + +### 14.1 Base settings + +`base.py` defines the shared application configuration. + +It contains: + +* database settings; +* Redis settings; +* Django cache settings; +* Celery settings; +* health checks; +* storage-provider variables; +* object-storage bucket configuration; +* logging; +* REST Framework configuration; +* rate-limiting settings; +* email settings; +* security defaults; +* application registration. + +### 14.2 Deployment settings + +`deployment.py` extends the base settings with: + +* mandatory database URL; +* persistent database connections; +* secure proxy handling; +* HTTPS redirection; +* secure cookies; +* HSTS; +* CORS configuration; +* deployment logging; +* optional Sentry; +* Celery Sentry integration; +* Redis Sentry integration; +* production-oriented template caching. + +### 14.3 Production and staging settings + +The production and staging modules are small wrappers around the deployment +configuration. + +### 14.4 Local settings + +The local settings contain development-specific overrides. + +### 14.5 Test settings + +The test settings contain test-specific configuration and service +substitutions. + +--- + +## 15. Database Configuration in Django + +The base database configuration is created through: + +```python +env.db("DATABASE_URL", default="postgres:///care") +``` + +The default database uses: + +```python +ATOMIC_REQUESTS = True +``` + +The base default for persistent connections is: + +```text +CONN_MAX_AGE=0 +``` + +The deployment settings override it with a default of: + +```text +CONN_MAX_AGE=60 +``` + +The database is the durable application system of record. + +The reviewed runtime does not use Redis as its principal durable database. + +--- + +## 16. Redis Configuration in Django + +The base settings define: + +```text +REDIS_URL +``` + +with a default of: + +```text +redis://localhost:6379 +``` + +Redis currently supports multiple runtime responsibilities. + +These responsibilities are described separately below. + +--- + +## 17. Redis as Django Cache + +The default Django cache is configured with: + +```text +django_redis.cache.RedisCache +``` + +The cache location is: + +```text +REDIS_URL +``` + +The client implementation is: + +```text +django_redis.client.DefaultClient +``` + +The cache configuration enables: + +```text +IGNORE_EXCEPTIONS=True +``` + +The application therefore treats many Redis cache failures as cache misses +instead of propagating the exception. + +--- + +## 18. Swagger Cache + +CARE defines a second cache named: + +```text +swagger_cache +``` + +This cache uses: + +```text +django.core.cache.backends.locmem.LocMemCache +``` + +Its location is: + +```text +swagger-schema-cache +``` + +The current runtime therefore already uses both: + +* Redis-backed shared cache; +* process-local memory cache. + +--- + +## 19. Redis as Celery Broker + +The Celery broker is configured through: + +```text +CELERY_BROKER_URL +``` + +Its default is: + +```text +REDIS_URL +``` + +Celery workers retrieve task messages from Redis. + +--- + +## 20. Redis as Celery Result Backend + +The Celery result backend is configured as: + +```python +CELERY_RESULT_BACKEND = CELERY_BROKER_URL +``` + +With the default configuration, Redis stores task-result state. + +--- + +## 21. Redis and Health Checks + +The health-check configuration includes a Celery queue-length check. + +This health check receives: + +```text +REDIS_URL +``` + +directly as its broker connection. + +The queue name is: + +```text +celery +``` + +The queue thresholds include: + +```text +info_length=50 +warning_length=0 +alert_length=200 +``` + +The zero warning threshold is documented in the code as a way to skip an +intermediate warning status. + +--- + +## 22. Redis and Report Progress + +Report-generation status is stored using Django's default cache. + +The report utility defines keys in the form: + +```text +report_generation_lock:_ +``` + +It stores integer progress values. + +The operations are: + +```python +cache.set(...) +cache.get(...) +cache.delete(...) +``` + +The default expiration is: + +```text +120 seconds +``` + +The code names these operations as locks, but the implementation is a cached +status/progress value rather than an atomic lock-acquisition mechanism. + +--- + +## 23. Rate Limiting + +The project includes: + +```text +django_ratelimit +``` + +in the installed applications. + +The base settings define: + +```text +DISABLE_RATELIMIT +RATE_LIMIT +``` + +The default configured rate is: + +```text +5/10m +``` + +The precise runtime behavior depends on the decorators and views using this +configuration. + +The rate-limit implementation may use Django's configured cache depending on +the call sites and library configuration. + +--- + +## 24. Celery Application + +Celery is initialized in: + +```text +config/celery_app.py +``` + +The module: + +* defaults `DJANGO_SETTINGS_MODULE` to `config.settings.production`; +* creates a Celery application named `care`; +* loads configuration from Django settings; +* uses the `CELERY_` namespace; +* sets `enable_utc=False`; +* uses the `Asia/Kolkata` timezone; +* autodiscovers tasks from installed Django applications. + +--- + +## 25. Celery Serialization and Limits + +The base settings define JSON-only task serialization: + +```text +CELERY_ACCEPT_CONTENT=["json"] +CELERY_TASK_SERIALIZER="json" +CELERY_RESULT_SERIALIZER="json" +``` + +The configured hard task limit is: + +```text +9000 seconds +``` + +The configured soft task limit is: + +```text +1800 seconds +``` + +--- + +## 26. Task Package + +The reviewed task package is: + +```text +care/emr/tasks/ +``` + +It contains: + +```text +__init__.py +cleanup_expired_token_slots.py +cleanup_incomplete_file_uploads.py +report_generation.py +totp.py +``` + +Three further task definitions exist outside this package: + +```text +care/emr/models/location.py:159 handle_cascade +care/emr/models/resource_category.py:123 summarise_monetary_components +care/emr/resources/account/sync_items.py:81 rebalance_account_task +``` + +These three are decorated with `@app.task` or `@shared_task` but are invoked as +ordinary function calls at nearly every call site, without `.delay()`. They +execute inline in the calling process. + +--- + +## 27. Periodic Task Registration + +Periodic tasks are registered in: + +```text +care/emr/tasks/__init__.py +``` + +The registration occurs through: + +```python +@current_app.on_after_finalize.connect +``` + +The callback calls: + +```python +sender.add_periodic_task(...) +``` + +Two schedules are currently registered. + +### 27.1 Expired token-slot cleanup + +The task runs daily at: + +```text +00:00 +``` + +### 27.2 Incomplete-upload cleanup + +The task runs every: + +```text +FILE_UPLOAD_EXPIRY_HOURS +``` + +converted into seconds. + +The default expiry value is: + +```text +24 hours +``` + +--- + +## 28. Expired Token-Slot Cleanup + +The task is: + +```text +cleanup_expired_token_slots +``` + +It is a Celery shared task. + +It: + +* logs the start of cleanup; +* queries `TokenSlot`; +* selects slots without related bookings; +* selects slots whose end time has passed; +* deletes the resulting queryset. + +The task accesses PostgreSQL through Django ORM. + +--- + +## 29. Incomplete File-Upload Cleanup + +The task is: + +```text +cleanup_incomplete_file_uploads +``` + +It: + +* calculates an expiration threshold; +* selects incomplete `FileUpload` records; +* processes records in pages of up to 1,000; +* deletes corresponding storage objects; +* deletes database records; +* repeats until no matching records remain. + +The task uses: + +```text +FileUpload.files_manager +``` + +for object deletion. + +The task accesses both: + +* PostgreSQL; +* S3-compatible object storage. + +--- + +## 30. Report Generation Task + +The task is: + +```text +generate_report_task +``` + +It receives: + +```text +template_id +report_type +associating_id +output_format +additional keyword arguments +``` + +It: + +1. Creates a report progress key. +2. Writes progress to Django cache. +3. Loads a `Template` through Django ORM. +4. Updates progress. +5. Generates and uploads the report. +6. Returns the resulting report-upload external ID. +7. Clears the progress key in a `finally` block. + +The task retries: + +```text +botocore.exceptions.ClientError +``` + +up to three times. + +The task expires after: + +```text +10 minutes +``` + +The task uses: + +* Django ORM; +* Django cache; +* report rendering code; +* object storage; +* Celery retry behavior; +* Celery result values. + +--- + +## 31. TOTP Email Tasks + +The task module defines: + +```text +send_totp_enabled_email +send_totp_disabled_email +``` + +Both are Celery shared tasks. + +Each task: + +* renders a Django email template; +* builds an HTML email; +* sends through Django's configured email backend; +* retries general exceptions; +* allows up to three retries; +* expires after ten minutes. + +--- + +## 32. Email Runtime + +The default email backend is: + +```text +django.core.mail.backends.smtp.EmailBackend +``` + +The configuration includes: + +```text +EMAIL_HOST +EMAIL_PORT +EMAIL_USER +EMAIL_PASSWORD +EMAIL_USE_TLS +EMAIL_FROM +``` + +Deployment settings enable TLS. + +The task system uses Django's email abstraction rather than a provider SDK +directly. + +--- + +## 33. Static Files + +Static files are configured through Django's `STORAGES` setting. + +The static-files backend is: + +```text +whitenoise.storage.CompressedManifestStaticFilesStorage +``` + +The static root is: + +```text +/staticfiles +``` + +The static URL is: + +```text +/staticfiles/ +``` + +The backend startup script runs: + +```bash +python manage.py collectstatic --noinput +``` + +The current runtime serves static files using WhiteNoise. + +--- + +## 34. Media Settings + +The base settings define: + +```text +MEDIA_ROOT +MEDIA_URL +``` + +The values point to: + +```text +care/media +/mediafiles/ +``` + +Clinical and facility object operations, however, are handled by the custom +S3-compatible file-manager implementation rather than Django's default storage +backend. + +--- + +## 35. Object-Storage Provider Configuration + +The base settings define: + +```text +BUCKET_PROVIDER +BUCKET_REGION +BUCKET_KEY +BUCKET_SECRET +BUCKET_ENDPOINT +BUCKET_EXTERNAL_ENDPOINT +BUCKET_HAS_FINE_ACL +``` + +The default provider value is: + +```text +aws +``` + +converted to uppercase. + +The provider value is validated against: + +```text +CSProvider +``` + +--- + +## 36. Declared Storage Providers + +The `CSProvider` enum includes: + +```text +AWS +AWS_ROLE_BASED +GCP +DIGITAL_OCEAN +MINIO +DOCKER +LOCAL +``` + +The provider enum is located in: + +```text +care/utils/csp/config.py +``` + +The enum identifies configured provider modes. + +The actual file operations remain implemented through an S3-compatible client. + +--- + +## 37. Storage Bucket Types + +CARE defines three logical bucket types: + +```text +PATIENT +FACILITY +REPORT +``` + +Reports currently use the same configured bucket as patient files. + +--- + +## 38. Patient File Bucket Configuration + +Patient files use the following settings for the bucket name and endpoints: + +```text +FILE_UPLOAD_BUCKET +FILE_UPLOAD_BUCKET_ENDPOINT +FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT +``` + +The credentials and region are **not** taken from the `FILE_UPLOAD_*` settings. + +`get_patient_bucket_config` in `care/utils/csp/config.py:46-56` reads: + +```text +FACILITY_S3_REGION +FACILITY_S3_KEY +FACILITY_S3_SECRET +``` + +The settings `FILE_UPLOAD_REGION`, `FILE_UPLOAD_KEY` and `FILE_UPLOAD_SECRET` +are defined at `config/settings/base.py:537-539` and are read by no code in the +repository. + +The local environment points these values to MinIO. + +--- + +## 39. Facility File Bucket Configuration + +Facility files use: + +```text +FACILITY_S3_BUCKET +FACILITY_S3_REGION +FACILITY_S3_KEY +FACILITY_S3_SECRET +FACILITY_S3_BUCKET_ENDPOINT +FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT +FACILITY_CDN +``` + +The setting names retain S3 terminology. + +--- + +## 40. Report Bucket Configuration + +Reports use the same bucket and endpoint configuration as patient files, and +therefore also read the `FACILITY_S3_*` credentials described in section 38. + +`get_report_bucket_config` is defined at `care/utils/csp/config.py:59-70`. + +The bucket type remains separately identified as: + +```text +REPORT +``` + +--- + +## 41. Internal and External Endpoints + +The storage configuration distinguishes between: + +```text +internal endpoint +external endpoint +``` + +The internal endpoint is used by server-side object operations. + +The external endpoint is used when generating client-facing signed URLs. + +In local Docker, the internal MinIO endpoint is reachable through the Docker +service name. + +The external endpoint is reachable from the host browser. + +--- + +## 42. Current Storage Client Configuration + +The bucket configuration helpers return boto3-style parameters: + +```text +region_name +aws_access_key_id +aws_secret_access_key +endpoint_url +``` + +For the `AWS_ROLE_BASED` provider mode, explicit credentials and endpoint +configuration are omitted. + +--- + +## 43. Current File Manager + +The current file manager is implemented in: + +```text +care/emr/utils/file_manager.py +``` + +The concrete class is: + +```text +S3FilesManager +``` + +It imports: + +```text +boto3 +botocore.exceptions.ClientError +``` + +For each operation, it creates a boto3 S3 client using the selected bucket +configuration. + +--- + +## 44. Object Key Format + +Objects are stored using: + +```text +/ +``` + +The key is built from fields on the associated CARE file object. + +--- + +## 45. Signed Upload URLs + +The current file manager implements: + +```text +signed_url +``` + +This method generates a pre-signed S3 URL for: + +```text +put_object +``` + +The generated URL allows a client to upload directly to the configured +S3-compatible bucket. + +The request parameters include: + +```text +Bucket +Key +ContentType +``` + +when a MIME type is available. + +The default expiration is: + +```text +one hour +``` + +--- + +## 46. Signed Download URLs + +The current file manager implements: + +```text +read_signed_url +``` + +It generates a pre-signed S3 URL for: + +```text +get_object +``` + +The response content disposition is selected according to MIME type. + +Selected images and PDF files use: + +```text +inline +``` + +Other files use: + +```text +attachment +``` + +The generated filename combines: + +```text +file name +file extension +``` + +The default expiration is one hour. + +--- + +## 47. Direct Object Upload + +The current file manager implements: + +```text +put_object +``` + +It calls the S3-compatible: + +```text +put_object +``` + +operation directly. + +The content is passed through the `Body` argument. + +Additional provider arguments can be supplied through keyword arguments. + +--- + +## 48. Direct Object Retrieval + +The current file manager implements: + +```text +get_object +``` + +It returns the provider response from: + +```text +get_object +``` + +The helper: + +```text +file_contents +``` + +reads the response body completely and returns: + +```text +content type +content bytes +``` + +--- + +## 49. Object Deletion + +The current file manager provides: + +```text +delete_object +delete_objects +``` + +Single deletion calls: + +```text +delete_object +``` + +Batch deletion calls: + +```text +delete_objects +``` + +The batch implementation handles a provider response with the error code: + +```text +NotImplemented +``` + +The current code explicitly identifies GCP as a provider for which batch +deletion may not be implemented through this compatibility path. + +--- + +## 50. Report Storage Flow + +The report-generation utility: + +1. Creates a `ReportUpload` database record. + +2. Generates report bytes. + +3. Calls: + + ```text + report_upload.files_manager.put_object + ``` + +4. Marks the upload as completed. + +5. Saves the database record. + +If object upload fails: + +* the `ReportUpload` record is deleted; +* the exception is propagated. + +--- + +## 51. File-Upload Expiration + +The base setting: + +```text +FILE_UPLOAD_EXPIRY_HOURS +``` + +defaults to: + +```text +24 +``` + +A value of zero disables the incomplete-upload cleanup schedule. + +--- + +## 52. File Validation Configuration + +The base settings define allowed MIME types for: + +* images; +* videos; +* audio; +* documents. + +They also define allowed file extensions and blocked executable or script +extensions. + +These settings are used as application-level file restrictions. + +--- + +## 53. Health-Check Configuration + +CARE uses: + +```text +healthy_django +``` + +The configured checks are: + +```text +Database +Cache +Celery Queue Length +``` + +### 53.1 Database check + +The database check uses the default Django database connection. + +### 53.2 Cache check + +The cache check uses the default Django cache. + +In the current base configuration, that cache is Redis-backed. + +### 53.3 Celery queue check + +The Celery check connects to the Redis broker and inspects the `celery` queue. + +--- + +## 54. Logging + +The base settings log to the console. + +The default formatter includes: + +```text +level +timestamp +module +process +thread +message +``` + +The deployment settings also use console logging. + +This is compatible with container-oriented log collection. + +--- + +## 55. Sentry Integration + +When: + +```text +SENTRY_DSN +``` + +is configured, the deployment settings initialize Sentry. + +Configured integrations include: + +```text +DjangoIntegration +CeleryIntegration +RedisIntegration +LoggingIntegration +``` + +Celery Beat monitoring is enabled in the Celery integration. + +--- + +## 56. Security and Proxy Configuration + +The deployment settings configure: + +```text +SECURE_PROXY_SSL_HEADER +SECURE_SSL_REDIRECT +SESSION_COOKIE_SECURE +CSRF_COOKIE_SECURE +SECURE_HSTS_SECONDS +SECURE_HSTS_INCLUDE_SUBDOMAINS +SECURE_HSTS_PRELOAD +SECURE_CONTENT_TYPE_NOSNIFF +``` + +CORS is configured through: + +```text +CORS_ALLOWED_ORIGINS +CORS_ALLOWED_ORIGIN_REGEXES +``` + +CSRF trusted origins are configured in the base settings. + +--- + +## 57. Authentication Runtime + +The REST API uses a combination of: + +```text +CustomJWTAuthentication +CustomBasicAuthentication +SessionAuthentication +TokenAuthentication +``` + +The default permissions require authenticated access and CARE-specific +authorization. + +The GCP runtime adaptation does not currently alter these application +authentication mechanisms. + +--- + +## 58. Plugin Configuration + +The application loads plugins through: + +```text +plug_config.manager +``` + +Plugin applications are appended to Django's installed applications. + +Plugin configuration is also loaded through the manager. + +Plugins may introduce additional runtime dependencies, tasks or application +behavior not listed in the core task package. + +--- + +## 59. Current External-Service Assumptions + +The base and deployment configurations include integrations or configuration +for: + +* SMTP; +* Amazon SNS-style SMS; +* Sentry; +* external Snowstorm FHIR service; +* object storage; +* Redis; +* PostgreSQL. + +Not all integrations are necessarily enabled in every deployment. + +--- + +## 60. Current Persistent Components + +The local runtime persists: + +```text +PostgreSQL data +Redis data +MinIO object data +database backups +``` + +The Django and Celery containers themselves are replaceable application +processes. + +--- + +## 61. Current Process Roles + +The runtime currently has two application process roles. + +### API role + +Identified through: + +```text +/tmp/container-role = api +``` + +Responsibilities include: + +* serving HTTP requests; +* Django application execution; +* static collection during startup. + +### Celery role + +Identified through: + +```text +/tmp/container-role = celery +``` + +Responsibilities include: + +* running migrations; +* synchronizing permissions; +* synchronizing value sets; +* executing background tasks; +* running Celery Beat. + +--- + +## 62. Current Runtime Coupling Summary + +The existing runtime contains the following direct couplings. + +| Concern | Current coupling | +| --------------------------- | ---------------------------------------------------------- | +| API startup | PostgreSQL and Redis | +| asynchronous task transport | Celery and Redis | +| task results | Celery and Redis | +| periodic scheduling | Celery Beat | +| default shared cache | Redis | +| report progress | Django default cache | +| object storage | boto3 and S3-compatible API | +| direct uploads | S3 pre-signed PUT URLs | +| direct downloads | S3 pre-signed GET URLs | +| local object storage | MinIO | +| application database | PostgreSQL | +| static files | WhiteNoise | +| health checks | database, Redis cache and Celery/Redis queue | +| monitoring | optional Sentry with Django, Celery and Redis integrations | + +--- + +## 63. Current Runtime Characteristics + +The current runtime is designed around continuously available service +processes. + +It assumes that: + +* PostgreSQL is continuously available; +* Redis is continuously available; +* a Celery worker is continuously running; +* Celery Beat is continuously running inside the worker; +* an S3-compatible object-storage service is available; +* the backend can wait for Redis before startup. + +The local architecture is suitable for Docker Compose and traditional +server-hosted deployment. + +--- + +## 64. Files Reviewed + +This inventory is based primarily on the following files: + +```text +README.md +Makefile +docker-compose.yaml +docker-compose.local.yaml +docker/.local.env +docker/.prebuilt.env +scripts/start-dev.sh +scripts/celery-dev.sh +config/celery_app.py +config/settings/base.py +config/settings/deployment.py +care/utils/csp/config.py +care/emr/utils/file_manager.py +care/emr/tasks/__init__.py +care/emr/tasks/cleanup_expired_token_slots.py +care/emr/tasks/cleanup_incomplete_file_uploads.py +care/emr/tasks/report_generation.py +care/emr/tasks/totp.py +care/emr/reports/report_utils.py +``` + +--- + +## 65. Additional Inventory Still Required + +A complete implementation inventory should additionally inspect: + +* every `.delay()` call; +* every `.apply_async()` call; +* every Celery task outside `care/emr/tasks`; +* task call sites introduced by plugins; +* whether API responses expose Celery task IDs; +* whether clients poll Celery results; +* every use of the default Django cache; +* every `django_ratelimit` decorator; +* every direct Redis import; +* every use of `files_manager`; +* frontend upload and download flows; +* health-check route definitions; +* production Dockerfiles; +* deployment workflows; +* GitHub Actions; +* existing Kubernetes or deployment manifests; +* plugin-specific storage behavior. + +--- + +## 66. Document Boundary + +This document describes the current runtime only. + +It intentionally does not decide: + +* whether Redis will remain mandatory; +* whether Cloud Tasks will replace Celery; +* whether Celery will remain available; +* whether storage will move to Django Storage API; +* whether direct-to-bucket uploads will be removed; +* how GCP services will be configured; +* how the migration will be sequenced. + +Those decisions belong in: + +```text +docs/xii/architecture/02-target-runtime.md +``` + +and supporting Architecture Decision Records. + +--- + +## 67. Next Document + +The next document is: + +```text +docs/xii/architecture/02-target-runtime.md +``` + +It will define the intended GCP runtime, including: + +* Cloud Run for the CARE API; +* Cloud SQL for PostgreSQL; +* Django Storage API and `django-storages`; +* uploads and downloads through Django; +* Cloud Storage; +* Cloud Tasks; +* private Cloud Run task execution; +* Cloud Scheduler; +* Cloud Run Jobs; +* optional Redis-compatible services; +* Secret Manager; +* Artifact Registry; +* IAM and service accounts; +* scaling and cost boundaries. diff --git a/docs/xii/architecture/02-target-runtime.md b/docs/xii/architecture/02-target-runtime.md new file mode 100644 index 0000000000..254703ff35 --- /dev/null +++ b/docs/xii/architecture/02-target-runtime.md @@ -0,0 +1,1914 @@ +--- +title: Target GCP Runtime +document: 02-target-runtime +version: 0.2.0 +status: Draft +source_repository: https://github.com/ohcnetwork/care +target_platform: Google Cloud Platform +--- + +# Target GCP Runtime + +## 1. Purpose + +This document defines the target production runtime for CARE on Google Cloud +Platform. + +It is based on the current runtime described in: + +```text +docs/xii/architecture/01-current-runtime.md +``` + +The objective is to run CARE without a permanently active virtual machine, +minimize idle infrastructure cost where practical, reuse PostgreSQL for +appropriate responsibilities, and preserve compatibility with the official +upstream repository. + +This document defines the desired end state. + +The detailed implementation sequence is documented in: + +```text +docs/xii/architecture/03-migration-plan.md +``` + +--- + +## 2. Primary Objective + +The target runtime SHALL replace the current continuously running +Docker-Compose-style production topology with managed and request-driven GCP +services. + +The default GCP architecture SHALL use: + +- Cloud Run for the CARE HTTP API; +- Cloud SQL for PostgreSQL; +- Cloud Storage for uploaded and generated files; +- Django Storage API through `django-storages`; +- Cloud Tasks for request-triggered asynchronous work; +- a private Cloud Run service for task execution; +- Cloud Scheduler for periodic triggers; +- Cloud Run Jobs for migrations, maintenance and batch workloads; +- Secret Manager for secrets; +- Artifact Registry for container images; +- Cloud Logging for container logs. + +PostgreSQL SHALL also be available as an optional shared backend for: + +- Django cache; +- task execution state; +- report progress; +- idempotency; +- rate limiting; +- transient shared state; +- distributed coordination; +- PostgreSQL-backed task queues in deployment profiles where their operational + trade-offs are acceptable. + +Redis-compatible storage SHALL remain optional for responsibilities that +benefit from lower-latency shared state. + +The target architecture SHALL NOT require: + +- a Compute Engine VM; +- self-managed PostgreSQL; +- self-managed MinIO; +- a permanently running Celery worker in the default GCP profile; +- a permanently running Celery Beat process; +- Redis as a mandatory production dependency; +- direct browser-to-bucket uploads; +- direct application use of `boto3` for CARE file storage. + +--- + +## 3. Design Principle: One Required Stateful Service + +PostgreSQL is already required as CARE's durable database. + +The architecture SHOULD reuse PostgreSQL for additional responsibilities when: + +- the workload is moderate; +- the responsibility benefits from shared state; +- the additional database load is acceptable; +- avoiding another managed service materially reduces cost or complexity; +- PostgreSQL provides suitable correctness and concurrency semantics. + +The system SHALL not introduce Redis merely because Redis is traditionally used +for a particular responsibility. + +Likewise, PostgreSQL SHALL not be used for every responsibility merely because +it is already available. + +Backend selection SHALL consider: + +- correctness; +- latency; +- throughput; +- contention; +- operational cost; +- connection usage; +- maintenance burden; +- scale-to-zero behavior. + +--- + +## 4. Target Architecture + +```mermaid +flowchart TD + CLIENT[CARE frontend and API clients] --> API[CARE API
Cloud Run] + + API --> SQL[(Cloud SQL
PostgreSQL)] + API --> STORAGE[(Cloud Storage
through django-storages)] + API --> TASKS[Cloud Tasks] + + TASKS --> WORKER[CARE Task Worker
Private Cloud Run] + WORKER --> SQL + WORKER --> STORAGE + + SCHEDULER[Cloud Scheduler] --> JOBS[Cloud Run Jobs] + SCHEDULER --> TASKS + + JOBS --> SQL + JOBS --> STORAGE + + SECRETS[Secret Manager] --> API + SECRETS --> WORKER + SECRETS --> JOBS + + REGISTRY[Artifact Registry] --> API + REGISTRY --> WORKER + REGISTRY --> JOBS + + REDIS[(Optional Redis-compatible service)] + API -. optional low-latency cache or rate limits .-> REDIS + WORKER -. optional transient state .-> REDIS + + PGQUEUE[Optional PostgreSQL-backed task queue] + API -. optional consolidated task backend .-> PGQUEUE + PGQUEUE -. consumed by optional worker .-> WORKER +``` + +The PostgreSQL-backed task queue shown in the diagram is optional. + +It is not part of the default scale-to-zero GCP profile. + +--- + +## 5. Supported Deployment Profiles + +The target architecture SHALL support multiple coherent profiles rather than a +single mandatory combination of services. + +### 5.1 Serverless GCP profile + +The recommended default profile is: + +```text +task backend: Cloud Tasks +cache backend: PostgreSQL or LocMem +rate-limit backend: PostgreSQL +transient-state backend: PostgreSQL +storage backend: Google Cloud Storage +scheduled work: Cloud Scheduler and Cloud Run Jobs +``` + +Properties: + +- no Redis required; +- API can scale to zero; +- task worker can scale to zero; +- no continuously polling queue worker; +- PostgreSQL remains the main permanent cost. + +### 5.2 PostgreSQL-consolidated profile + +An optional consolidated profile MAY use: + +```text +task backend: PostgreSQL-backed queue +cache backend: PostgreSQL +rate-limit backend: PostgreSQL +transient-state backend: PostgreSQL +storage backend: Google Cloud Storage or S3-compatible storage +``` + +Properties: + +- no Redis required; +- no Cloud Tasks required; +- fewer infrastructure products; +- task queue and cache increase PostgreSQL load; +- a queue worker must remain available or be invoked periodically; +- immediate tasks may require a continuously running worker; +- it does not provide the same scale-to-zero behavior as Cloud Tasks. + +This profile is primarily appropriate for: + +- traditional container deployments; +- on-premise deployments; +- low-volume deployments; +- environments prioritizing service consolidation; +- installations willing to keep a worker active; +- environments where Cloud Tasks is unavailable or undesirable. + +### 5.3 Redis-optimized profile + +An optional Redis-compatible profile MAY use: + +```text +task backend: Cloud Tasks or Celery +cache backend: Redis +rate-limit backend: Redis +transient-state backend: Redis +storage backend: Google Cloud Storage or S3-compatible storage +``` + +Possible providers include: + +- Upstash; +- Google Memorystore; +- Redis; +- Valkey; +- Dragonfly; +- other compatible services. + +### 5.4 Traditional CARE profile + +The traditional profile MAY continue to use: + +```text +PostgreSQL +Redis +Celery +Celery Beat +MinIO or S3 +``` + +This remains useful for local development and conventional server +deployments. + +--- + +## 6. Runtime Components + +The default GCP runtime contains: + +```text +care-api +care-worker +care-jobs +``` + +All three SHOULD use the same container image. + +They differ by: + +- startup command; +- IAM policy; +- scaling settings; +- runtime purpose; +- exposed routes; +- resource allocation. + +An optional PostgreSQL queue profile MAY additionally run: + +```text +care-queue-worker +``` + +This process consumes tasks from PostgreSQL. + +It MAY use the same image, but it has different lifecycle requirements from the +request-driven Cloud Tasks worker. + +--- + +## 7. CARE API Service + +The CARE API SHALL run as a Cloud Run service. + +Its responsibilities include: + +- serving REST API requests; +- authenticating users; +- enforcing CARE permissions; +- receiving file uploads; +- returning file downloads; +- reading and writing PostgreSQL data; +- dispatching asynchronous work; +- serving health and readiness endpoints; +- serving static assets through WhiteNoise. + +The API service SHALL NOT: + +- run Celery Beat; +- execute scheduled cleanup loops; +- run database migrations during every startup; +- expose object-storage credentials to the frontend; +- generate direct upload URLs for browsers; +- require Redis unless Redis-backed functionality is selected. + +The API MAY enqueue work into: + +- Cloud Tasks; +- Celery; +- a PostgreSQL-backed task queue; + +according to `CARE_TASK_BACKEND`. + +--- + +## 8. Cloud Run Scaling + +The API SHOULD initially use: + +```text +minimum instances: 0 +``` + +unless operational requirements justify warm instances. + +The API SHALL define a maximum instance count based on: + +- expected traffic; +- Cloud SQL connection limits; +- cost controls; +- per-instance concurrency; +- memory requirements; +- storage streaming behavior. + +The following values MUST be designed together: + +```text +Cloud Run concurrency +maximum Cloud Run instances +Gunicorn worker count +Gunicorn thread count +database connection lifetime +Cloud SQL maximum connections +database-backed cache traffic +database-backed rate-limit traffic +database-backed queue traffic +``` + +Using PostgreSQL for additional responsibilities increases the importance of +conservative connection and query management. + +--- + +## 9. HTTP Server + +The GCP runtime SHALL use a production WSGI server. + +The Django development server and `runserver_plus` SHALL NOT be used in +production. + +The expected process is conceptually: + +```bash +gunicorn config.wsgi:application \ + --bind "0.0.0.0:${PORT}" \ + --workers "${GUNICORN_WORKERS}" \ + --threads "${GUNICORN_THREADS}" \ + --timeout "${GUNICORN_TIMEOUT}" +``` + +The process MUST listen on the `PORT` environment variable supplied by Cloud +Run. + +Exact values SHALL be determined through load testing and database connection +limits. + +--- + +## 10. Cloud SQL + +PostgreSQL SHALL remain the durable system of record. + +Cloud SQL for PostgreSQL SHALL replace local or self-managed PostgreSQL in GCP +environments. + +The application SHALL continue using Django ORM normally. + +No repository abstraction or replacement persistence layer is required. + +### 10.1 Core responsibilities + +Cloud SQL SHALL store: + +- CARE domain records; +- authentication and authorization data; +- audit data; +- durable task execution state; +- idempotency records; +- report-generation progress when PostgreSQL is selected; +- shared transient state when PostgreSQL is selected; +- rate-limit counters when PostgreSQL is selected; +- Django database-cache entries when PostgreSQL is selected; +- PostgreSQL-backed task queue records when that backend is selected. + +### 10.2 Connectivity + +The API, worker and jobs SHALL connect using the supported Cloud Run and Cloud +SQL integration mechanism. + +Database credentials SHALL be injected through Secret Manager or equivalent +runtime configuration. + +The database SHALL NOT be exposed through unrestricted public networking. + +### 10.3 Connection management + +The GCP settings SHALL explicitly configure: + +```text +CONN_MAX_AGE +CONN_HEALTH_CHECKS +``` + +where supported by the installed Django version. + +The design SHALL account for: + +```text +service instances × processes × threads × database aliases +``` + +Database-backed caching and queuing SHALL not create uncontrolled independent +connection pools. + +### 10.4 Workload isolation + +The initial implementation MAY use the same Cloud SQL database for: + +- CARE application data; +- cache tables; +- rate-limit tables; +- transient-state tables; +- task queue tables. + +Production deployments SHOULD monitor contention and database growth. + +High-volume deployments MAY isolate infrastructure tables using: + +- a separate PostgreSQL schema; +- a separate database in the same Cloud SQL instance; +- a separate Cloud SQL instance when operationally justified. + +### 10.5 Availability + +Development environments MAY use a low-cost, non-HA configuration. + +Production SHALL document: + +- availability requirements; +- backups; +- point-in-time recovery; +- maintenance windows; +- deletion protection; +- restore procedures; +- retention periods. + +Cloud SQL is expected to represent the principal permanent baseline cost. + +--- + +## 11. Storage Architecture + +**Django Storage API is the architecture.** It is the single application-level +abstraction for object persistence. Providers are implementation details behind +it, selected by configuration alone. + +Stated explicitly, because these are four separate claims and only the third and +fourth concern GCP: + +1. **Django Storage API is the architecture** — not S3, not GCS, not MinIO. +2. **MinIO through `storages.backends.s3.S3Storage` is the default local + profile.** `CARE_STORAGE_BACKEND` defaults to `s3`, so a local checkout keeps + working with no configuration change and no GCP value of any kind. +3. **Generic S3-compatible storage remains supported** — AWS S3, MinIO and other + providers `django-storages` supports, as a first-class deployment profile + rather than a legacy path. +4. **GCS is the initial GCP storage profile, not the only supported provider.** + It is one implementation of the abstraction; adding another provider is a + settings change, not an application change. + +**Implemented in IS-01.** `config/storage.py` builds the aliases; +`config/settings/base.py` selects the backend. + +All CARE file storage SHALL use Django's Storage API. + +The implementation SHALL use: + +```text +django-storages +``` + +The application SHALL NOT maintain independent manual clients for S3, MinIO or +GCS. + +The application SHALL NOT use `boto3` directly for CARE file persistence after +migration. + +The application SHALL NOT use `google-cloud-storage` directly outside the +storage backend implementation. + +**Status: all three hold.** No provider client is constructed anywhere for object +persistence *or* transport, and nothing imports `google-cloud-storage` at all. + +**Object transport is entirely mediated by CARE.** No application code generates +a storage-provider URL and no client receives one. Presigned upload and download +are removed, as are the unsigned bucket URLs that served facility cover images +and user avatars. Every bucket can be private. `care/utils/csp/` — the +provider-credential and endpoint resolver that existed for signed URLs — is +deleted. + +One provider-specific reference remains, outside persistence and transport: +`care/emr/tasks/report_generation.py` imports `botocore`'s `ClientError` for +Celery retry configuration. It constructs no client and performs no storage +operation, but it will not fire under `gcs`. See +`inventory/unresolved-items.md` S2. + +**This is a gap in the target runtime, not merely an inventory note.** Retry +configuration that names a provider exception type is provider-specific code by +another route: under `gcs` a transient upload failure raises +`google.api_core.exceptions.*`, no retry fires, and the report fails on its +first attempt with no signal that a retry policy was ever intended. + +The target runtime SHALL therefore satisfy one of: + +- the storage boundary raises a provider-neutral exception type that retry + policies name, so a transient failure retries identically under either + backend; **or** +- report generation is excluded from the set of components declared + production-ready under `gcs`, and that exclusion is stated wherever readiness + is claimed. + +Until one holds, `gcs` SHALL NOT be described as production-ready for report +generation. Both options are outside IS-01's remit: ES-01 §31 forbids modifying +Celery, and the first option changes behaviour under `s3` as well. + +See `inventory/storage-call-sites.md` §11 for the per-call-site record. + +--- + +## 12. Storage Backends + +The target runtime SHALL support provider selection through Django's +`STORAGES` setting. + +Principal backends: + +```text +storages.backends.s3.S3Storage +storages.backends.gcloud.GoogleCloudStorage +``` + +The S3 backend supports: + +- AWS S3; +- MinIO; +- compatible S3 providers supported by `django-storages`. + +The GCS backend supports: + +- native Google Cloud Storage; +- Application Default Credentials; +- Cloud Run service-account identity. + +--- + +## 13. Logical Storage Aliases + +CARE SHALL use separate aliases for: + +```text +patient +facility +report +staticfiles +``` + +They SHALL be retrieved through: + +```python +from django.core.files.storage import storages +``` + +Example: + +```python +patient_storage = storages["patient"] +facility_storage = storages["facility"] +report_storage = storages["report"] +``` + +The aliases SHALL remain stable regardless of provider. + +--- + +## 14. GCP Storage Configuration + +The GCP settings SHALL configure aliases similar to: + +```python +STORAGES = { + "patient": { + "BACKEND": "storages.backends.gcloud.GoogleCloudStorage", + "OPTIONS": { + "bucket_name": env("CARE_PATIENT_STORAGE_BUCKET"), + "project_id": env("GCP_PROJECT_ID"), + }, + }, + "facility": { + "BACKEND": "storages.backends.gcloud.GoogleCloudStorage", + "OPTIONS": { + "bucket_name": env("CARE_FACILITY_STORAGE_BUCKET"), + "project_id": env("GCP_PROJECT_ID"), + }, + }, + "report": { + "BACKEND": "storages.backends.gcloud.GoogleCloudStorage", + "OPTIONS": { + "bucket_name": env("CARE_REPORT_STORAGE_BUCKET"), + "project_id": env("GCP_PROJECT_ID"), + }, + }, + "staticfiles": { + "BACKEND": ( + "whitenoise.storage." + "CompressedManifestStaticFilesStorage" + ), + }, +} +``` + +The exact options SHALL follow the installed `django-storages` version. + +Cloud Run SHALL use service-account credentials rather than committed JSON +keys. + +--- + +## 15. Local Storage Configuration + +Local development SHALL continue supporting MinIO through: + +```text +storages.backends.s3.S3Storage +``` + +with a custom endpoint. + +Equivalent aliases SHALL exist for patient, facility and report files. + +Local contributors SHALL not require GCP credentials. + +--- + +## 16. Upload Policy + +All uploads SHALL pass through the CARE Django API. + +The API SHALL: + +1. authenticate and authorize the caller; +2. validate file metadata; +3. validate extension; +4. validate MIME type; +5. enforce size limits; +6. save through the selected Django storage alias; +7. persist or update the corresponding record; +8. return a CARE-level response. + +The frontend SHALL NOT: + +- request a signed upload URL; +- upload directly to Cloud Storage; +- upload directly to MinIO; +- upload directly to S3; +- receive storage credentials; +- select the provider. + +--- + +## 17. Download Policy + +Downloads SHALL pass through CARE. + +The API SHALL: + +1. authenticate the caller; +2. authorize access; +3. determine the storage alias and object name; +4. open the object through Django Storage API; +5. return a streaming response; +6. set safe content headers. + +Direct provider URLs SHALL not be the normal download mechanism. + +--- + +## 18. Streaming and Temporary Files + +The application SHALL avoid loading complete files into memory when streaming +is possible. + +Uploads SHALL use Django upload handlers. + +The deployment SHALL define: + +```text +DATA_UPLOAD_MAX_MEMORY_SIZE +FILE_UPLOAD_MAX_MEMORY_SIZE +``` + +and CARE-specific maximum file sizes. + +Downloads SHALL use streaming responses. + +Very large media workflows are outside the initial scope. + +--- + +## 19. Storage Consistency + +Database and object storage cannot share one atomic transaction. + +CARE SHALL manage partial failures explicitly. + +Cleanup mechanisms SHALL handle: + +- orphaned objects; +- incomplete records; +- failed deletions; +- interrupted uploads. + +--- + +## 20. Static Files + +Static files SHALL remain served through WhiteNoise. + +Static collection SHALL happen during image build: + +```bash +python manage.py collectstatic --noinput +``` + +Cloud Storage SHALL not be required for static files. + +--- + +## 21. Task Backend Architecture + +CARE SHALL support configurable task backends. + +The initial supported values SHOULD be: + +```text +cloud_tasks +postgres +celery +``` + +Conceptually: + +```text +CARE_TASK_BACKEND=cloud_tasks +CARE_TASK_BACKEND=postgres +CARE_TASK_BACKEND=celery +``` + +### 21.1 Cloud Tasks + +Default for GCP serverless deployments. + +### 21.2 PostgreSQL task queue + +Optional for consolidated or traditional deployments. + +It SHALL use a task queue intentionally designed around PostgreSQL. + +It SHALL NOT use an obsolete or unsupported Celery database broker transport. + +A suitable implementation MAY use a maintained PostgreSQL-backed task library +with: + +- Django integration; +- row locking; +- retries; +- scheduled execution; +- concurrency control; +- task status; +- worker health checks. + +### 21.3 Celery + +Retained for local and traditional deployments. + +--- + +## 22. Task Dispatch Contract + +The task-dispatch API SHALL conceptually support: + +```python +enqueue_task( + task_name, + payload, + delay_seconds=None, + task_id=None, +) +``` + +It SHALL return an external task identifier. + +Payloads SHALL be JSON-serializable. + +Tasks SHOULD receive opaque identifiers and load state from PostgreSQL. + +The application SHALL NOT enqueue: + +- model instances; +- open files; +- lazy querysets; +- credentials; +- provider clients; +- complete clinical records when identifiers suffice. + +--- + +## 23. Cloud Tasks Backend + +The Cloud Tasks implementation SHALL: + +- enqueue HTTP-target tasks; +- target a private Cloud Run worker; +- use OIDC authentication; +- support delayed delivery; +- serialize JSON payloads; +- return the Cloud Tasks name; +- configure retries through infrastructure; +- avoid static credentials. + +This is the preferred GCP backend when scale-to-zero behavior is important. + +--- + +## 24. PostgreSQL Task Backend + +The PostgreSQL backend MAY enqueue tasks into Cloud SQL. + +It SHALL use PostgreSQL-native transactional and locking semantics. + +Potential benefits include: + +- no separate broker service; +- transactional task creation; +- durable task records; +- simpler local and on-premise operation; +- easier inspection through SQL; +- consolidation of backups and monitoring. + +Trade-offs include: + +- additional database load; +- table growth and cleanup requirements; +- lock contention; +- increased connection usage; +- competition with clinical workloads; +- need for an active queue consumer; +- reduced scale-to-zero behavior. + +### 24.1 Worker lifecycle + +A PostgreSQL-backed queue does not execute tasks by itself. + +It requires a worker that: + +- waits through polling or PostgreSQL notifications; +- claims available jobs; +- runs handlers; +- updates job status; +- retries failures. + +For immediate execution, the worker generally must remain active. + +Therefore, the PostgreSQL backend SHALL NOT be described as equivalent to Cloud +Tasks for serverless scale-to-zero operation. + +### 24.2 GCP execution options + +A PostgreSQL queue worker MAY run as: + +- a Cloud Run service with at least one active instance; +- a traditional container worker; +- a Kubernetes worker; +- a VM or on-premise process; +- a periodically invoked Cloud Run Job for non-urgent batches. + +The periodically invoked job model is only appropriate when queue latency is +allowed to match the invocation schedule. + +### 24.3 Transactional enqueue + +Where supported, enqueueing a task inside the same PostgreSQL transaction as a +domain update MAY ensure that: + +- both the domain change and task creation commit; +- or neither commits. + +This is a useful option for work coupled to database state. + +It does not remove the need for idempotent task execution. + +--- + +## 25. Celery Backend + +Celery SHALL preserve existing local behavior. + +Local deployments MAY continue using Redis as: + +```text +broker +result backend +``` + +Celery Beat MAY continue locally. + +The GCP default deployment SHALL not start Celery when another task backend is +selected. + +--- + +## 26. Worker Services + +### 26.1 Cloud Tasks worker + +The default GCP worker SHALL be a private Cloud Run HTTP service. + +It SHALL scale to zero. + +It SHALL accept only authenticated task requests. + +### 26.2 PostgreSQL queue worker + +The optional PostgreSQL worker SHALL consume jobs from PostgreSQL. + +It SHALL not need a public HTTP endpoint. + +It MAY require a continuously active process. + +### 26.3 Celery worker + +The traditional worker SHALL continue consuming from its configured Celery +broker. + +All worker types SHOULD invoke the same reusable task logic. + +--- + +## 27. Task Handler Registration + +Task handlers SHALL be explicitly registered. + +The application SHALL NOT execute arbitrary Python paths supplied in a +payload. + +Only server-defined names SHALL be executable. + +--- + +## 28. Reusable Task Logic + +Existing Celery tasks SHALL be refactored when necessary into: + +```text +reusable function +thin Celery wrapper +thin Cloud Tasks handler +thin PostgreSQL queue wrapper +``` + +Business behavior SHALL not be duplicated per backend. + +--- + +## 29. Task Result Handling + +Task results that matter to CARE SHALL be persisted in: + +- existing domain records; +- report-upload records; +- task execution records; +- PostgreSQL queue job records where selected; +- other explicit application state. + +Cloud Tasks response bodies SHALL not be treated as a durable result backend. + +Redis task results SHALL not be mandatory. + +--- + +## 30. Task Idempotency + +All task backends SHALL be treated as capable of retrying or redelivering work. + +Idempotency SHALL use: + +- database state; +- unique constraints; +- conditional updates; +- object existence; +- idempotency keys; +- task-execution tables where required. + +Redis SHALL not be the sole guarantee of clinical consistency. + +--- + +## 31. Task Classification + +### Cloud Tasks + +Appropriate for: + +- request-triggered work; +- email delivery; +- bounded report generation; +- serverless GCP execution; +- work requiring prompt dispatch. + +### PostgreSQL task queue + +Appropriate for: + +- low-to-moderate volume; +- consolidated infrastructure; +- transactional enqueue; +- traditional or always-on workers; +- deployments without a managed queue service. + +### Cloud Scheduler and Cloud Run Jobs + +Appropriate for: + +- periodic cleanup; +- migrations; +- bulk processing; +- maintenance; +- delayed batch execution. + +### Celery + +Appropriate for: + +- existing local behavior; +- traditional Redis/RabbitMQ deployments; +- compatibility with upstream. + +--- + +## 32. Initial Task Mapping + +The default GCP mapping is: + +| Current task | Default GCP mechanism | Optional PostgreSQL mechanism | +|---|---|---| +| TOTP enabled email | Cloud Tasks | PostgreSQL queue | +| TOTP disabled email | Cloud Tasks | PostgreSQL queue | +| report generation | Cloud Tasks | PostgreSQL queue | +| expired token cleanup | Scheduler + Cloud Run Job | Scheduled PostgreSQL task or management command | +| incomplete upload cleanup | Scheduler + Cloud Run Job | Scheduled PostgreSQL task or management command | + +The implementation SHALL validate duration, idempotency and call sites. + +--- + +## 33. Cloud Scheduler + +Cloud Scheduler SHALL replace Celery Beat in the default GCP profile. + +Schedules SHALL be committed through Terraform or another deployment +configuration. + +Cloud Scheduler MAY: + +- invoke a Cloud Run Job; +- enqueue a Cloud Task; +- invoke a protected management endpoint when justified. + +A PostgreSQL queue deployment MAY instead use its queue library's periodic-task +support, but only when it intentionally runs an active worker. + +--- + +## 34. Cloud Run Jobs + +Cloud Run Jobs SHALL execute: + +```text +database migrations +permission synchronization +value-set synchronization +fixtures +bulk cleanup +scheduled maintenance +batch imports +``` + +The same CARE image SHOULD be reused. + +Migrations SHALL not run during ordinary API startup. + +--- + +## 35. Cache Backend Architecture + +CARE SHALL support configurable Django cache backends. + +The initial supported values SHOULD include: + +```text +postgres +locmem +redis +dummy +``` + +Conceptual selection: + +```text +CARE_CACHE_BACKEND=postgres +CARE_CACHE_BACKEND=locmem +CARE_CACHE_BACKEND=redis +CARE_CACHE_BACKEND=dummy +``` + +--- + +## 36. PostgreSQL Database Cache + +PostgreSQL MAY be the default shared cache for the low-cost GCP profile. + +It SHALL use Django's database cache backend: + +```python +CACHES = { + "default": { + "BACKEND": "django.core.cache.backends.db.DatabaseCache", + "LOCATION": "care_cache", + "OPTIONS": { + "MAX_ENTRIES": 10000, + "CULL_FREQUENCY": 3, + }, + }, +} +``` + +The deployment SHALL create the cache table explicitly: + +```bash +python manage.py createcachetable +``` + +### 36.1 Appropriate uses + +Database cache is suitable for: + +- shared cache values across Cloud Run instances; +- low-to-moderate cache traffic; +- report progress; +- regenerated configuration; +- rate-limit support where semantics are compatible; +- avoiding an additional Redis service. + +### 36.2 Trade-offs + +Database caching adds: + +- database reads and writes; +- table bloat; +- expiration cleanup work; +- contention with application queries; +- persistent storage use; +- latency compared with in-memory cache. + +It SHALL not be assumed to provide Redis-level latency or throughput. + +### 36.3 Cache is not the source of truth + +Values stored through Django's cache API SHALL remain disposable. + +Durable application state SHALL use normal models, even when both are stored in +PostgreSQL. + +--- + +## 37. LocMem Cache + +LocMem MAY be used for: + +- instance-local performance optimizations; +- Swagger schemas; +- regenerated data; +- non-shared values. + +It SHALL be treated as: + +- process-local; +- ephemeral; +- non-coordinated; +- unsuitable for globally consistent state. + +--- + +## 38. Redis Cache + +Redis MAY be used for: + +- high-frequency shared cache; +- lower-latency counters; +- larger shared transient workloads; +- deployments where PostgreSQL cache load becomes excessive. + +The implementation SHOULD use standard Redis-compatible URLs. + +Upstash MAY be selected without application-specific Upstash coupling. + +--- + +## 39. Cache Backend Selection Guidance + +| Requirement | Recommended backend | +|---|---| +| no shared cache needed | LocMem | +| shared, moderate-volume cache with minimum services | PostgreSQL | +| high-frequency, low-latency shared cache | Redis-compatible | +| tests with caching disabled | Dummy | +| durable business state | normal PostgreSQL models, not cache | + +The backend SHALL be selected through configuration, not provider conditionals +spread through application code. + +--- + +## 40. Report Progress + +Report progress SHALL use one of: + +```text +PostgreSQL database cache +dedicated PostgreSQL model +Redis-compatible cache +``` + +The default low-cost profile MAY use the PostgreSQL database cache. + +A dedicated model SHOULD be preferred when progress must be: + +- durable; +- auditable; +- queryable after expiration; +- associated with task execution history. + +Report progress SHALL remain status information, not an integrity lock. + +--- + +## 41. Rate Limiting + +Globally consistent rate limits SHALL use shared state. + +Supported options include: + +```text +PostgreSQL +Redis-compatible storage +``` + +LocMem SHALL not be used for global enforcement across Cloud Run instances. + +PostgreSQL rate limiting MAY use: + +- dedicated models; +- atomic updates; +- database constraints; +- short-lived counter rows. + +Cloud Armor MAY complement application limits, but SHALL not replace +user-specific or workflow-specific policies. + +--- + +## 42. Shared Transient State + +Shared transient state MAY use: + +```text +PostgreSQL models +Django DatabaseCache +Redis-compatible storage +``` + +Selection depends on whether state must be: + +- durable; +- queryable; +- audited; +- high-frequency; +- automatically expired. + +Transient values that affect clinical correctness SHOULD use explicit +PostgreSQL models rather than a disposable cache backend. + +--- + +## 43. Distributed Coordination + +PostgreSQL MAY support coordination through: + +- row-level locks; +- `SELECT ... FOR UPDATE`; +- unique constraints; +- conditional updates; +- advisory locks; +- queue-library locking mechanisms. + +Redis MAY support optional short-lived distributed locks. + +Neither cache locks nor queue locks SHALL replace database constraints for +clinical integrity. + +--- + +## 44. Default Redis-Free Profile + +The recommended initial Redis-free profile SHOULD be: + +```text +CARE_TASK_BACKEND=cloud_tasks +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +``` + +LocMem MAY remain configured as a separate cache alias for local, +non-coordinated optimizations. + +This profile requires no `REDIS_URL`. + +--- + +## 45. Fully Consolidated PostgreSQL Profile + +An optional profile MAY use: + +```text +CARE_TASK_BACKEND=postgres +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +``` + +This minimizes infrastructure products but requires: + +- PostgreSQL queue tables; +- an active queue worker; +- monitoring of database pressure; +- cleanup of cache and queue records; +- explicit capacity planning. + +It SHALL not be presented as a fully serverless profile. + +--- + +## 46. Optional Redis Profile + +A Redis-enabled profile MAY use: + +```text +CARE_CACHE_BACKEND=redis +CARE_RATE_LIMIT_BACKEND=redis +CARE_TRANSIENT_STATE_BACKEND=redis +``` + +Connection variables SHOULD remain separate: + +```text +REDIS_CACHE_URL +REDIS_RATE_LIMIT_URL +REDIS_TRANSIENT_STATE_URL +``` + +They MAY point to the same instance. + +They SHALL not be required to do so. + +--- + +## 47. Upstash Compatibility + +Upstash MAY be used through the standard Redis protocol. + +Configuration SHALL remain provider-neutral: + +```text +CARE_CACHE_BACKEND=redis +REDIS_CACHE_URL=rediss://... +``` + +The application SHOULD NOT require: + +```text +USE_UPSTASH=true +``` + +Sensitive values SHALL be minimized and reviewed before storage in an external +Redis-compatible service. + +--- + +## 48. Health Checks + +Mandatory GCP checks SHALL include: + +```text +application process +PostgreSQL +``` + +Optional checks MAY include: + +```text +selected cache backend +selected queue backend +required external dependencies +``` + +Health checks SHALL reflect configured backends. + +Examples: + +- Cloud Tasks profile: no Celery queue check; +- PostgreSQL queue profile: check queue schema and database connectivity; +- Redis profile: check Redis only when Redis functionality is required; +- database-cache profile: check PostgreSQL and cache table availability. + +--- + +## 49. Logging + +API, worker and jobs SHALL log to stdout and stderr. + +Structured logs SHOULD include: + +```text +severity +timestamp +service +revision +request ID +task ID +task backend +task name +duration +retry count +status +``` + +Logs SHALL not contain sensitive clinical payloads or credentials. + +--- + +## 50. Sentry + +Sentry MAY remain optional. + +Integrations SHALL match selected backends. + +Examples: + +- Celery integration only when Celery is active; +- Redis integration only when Redis is active; +- Django integration for API and HTTP task worker; +- normal exception capture for PostgreSQL queue workers and jobs. + +--- + +## 51. Secret Manager + +Secret Manager SHALL store: + +```text +DJANGO_SECRET_KEY +database credentials +email credentials +Sentry DSN +JWT or JWKS material +optional Redis URLs +external-service credentials +``` + +PostgreSQL cache and queue backends SHOULD reuse the normal database identity +where appropriate rather than introduce separate secrets unnecessarily. + +--- + +## 52. Service Accounts and IAM + +Separate service accounts SHOULD exist for: + +```text +care-api +care-worker +care-jobs +care-tasks-invoker +care-deployer +``` + +A PostgreSQL queue worker MAY use the worker identity but does not require Cloud +Run invocation permissions unless it also exposes HTTP routes. + +Buckets SHALL not be public. + +Cloud Tasks worker invocation SHALL require IAM authentication. + +--- + +## 53. Artifact Registry and Image + +Artifact Registry SHALL store immutable images tagged with: + +```text +Git commit SHA +release tag +``` + +The same image SHOULD support: + +```text +API command +Cloud Tasks worker command +PostgreSQL queue worker command +Cloud Run Job commands +Celery worker command +``` + +The image SHALL not embed secrets or run migrations in its default entrypoint. + +--- + +## 54. GCP Settings Module + +The repository SHALL add: + +```text +config/settings/gcp.py +``` + +It SHALL inherit from: + +```python +from .deployment import * +``` + +It SHALL configure: + +- Cloud Run behavior; +- Cloud SQL; +- Django storage aliases; +- task backend; +- cache backend; +- rate-limit backend; +- transient-state backend; +- health checks; +- logging; +- optional Redis; +- upload limits. + +It SHALL not heavily rewrite `deployment.py`. + +--- + +## 55. Environment Variables + +Core variables SHOULD include: + +```text +DJANGO_SETTINGS_MODULE=config.settings.deployment + +GCP_PROJECT_ID +GCP_REGION + +CARE_TASK_BACKEND=cloud_tasks|postgres|celery +CARE_CACHE_BACKEND=postgres|locmem|redis|dummy +CARE_RATE_LIMIT_BACKEND=postgres|redis +CARE_TRANSIENT_STATE_BACKEND=postgres|redis + +CARE_PATIENT_STORAGE_BUCKET +CARE_FACILITY_STORAGE_BUCKET +CARE_REPORT_STORAGE_BUCKET +``` + +Cloud Tasks variables: + +```text +GCP_TASKS_LOCATION +GCP_TASKS_QUEUE +GCP_WORKER_URL +GCP_TASKS_SERVICE_ACCOUNT +``` + +PostgreSQL queue variables MAY include: + +```text +CARE_POSTGRES_QUEUE_SCHEMA +CARE_POSTGRES_QUEUE_NAMES +CARE_POSTGRES_WORKER_CONCURRENCY +CARE_POSTGRES_WORKER_POLL_INTERVAL +``` + +Optional Redis: + +```text +REDIS_CACHE_URL +REDIS_RATE_LIMIT_URL +REDIS_TRANSIENT_STATE_URL +``` + +--- + +## 56. Frontend Impact + +The frontend SHALL: + +- upload files to CARE endpoints; +- download files from CARE endpoints; +- stop requesting direct-upload URLs; +- stop sending files directly to buckets; +- stop depending on storage-provider responses. + +No frontend change is required merely because cache or task state moves between +PostgreSQL and Redis. + +--- + +## 57. Local Development + +Docker Compose SHALL continue providing: + +```text +PostgreSQL +Redis +MinIO +Celery +Celery Beat +Django backend +``` + +The project SHOULD additionally support testing the consolidated PostgreSQL +profile locally. + +Local tests SHOULD cover: + +```text +Celery + Redis +Cloud Tasks dispatcher through mocks or emulator strategy +PostgreSQL cache +PostgreSQL task queue, when implemented +MinIO through django-storages +``` + +--- + +## 58. Cost Boundaries + +### Can scale to zero + +```text +CARE API +Cloud Tasks HTTP worker +Cloud Run Jobs +``` + +### May require an active process + +```text +PostgreSQL task queue worker +Celery worker +Celery Beat +``` + +### Usage-based + +```text +Cloud Tasks +Cloud Scheduler +Cloud Storage operations +Artifact Registry +Secret Manager +Cloud Logging +``` + +### Persistent baseline cost + +```text +Cloud SQL +stored Cloud Storage data +optional managed Redis +``` + +Using PostgreSQL for cache and queues may reduce the number of services but may +require a larger Cloud SQL instance. + +The deployment SHALL compare total cost rather than only counting services. + +--- + +## 59. Security Requirements + +The target runtime SHALL enforce: + +- private buckets; +- no direct frontend credentials; +- authenticated task invocation; +- no static production service-account keys; +- controlled Cloud SQL access; +- minimal task payloads; +- no clinical data in logs; +- least-privilege IAM; +- encrypted transport; +- separate environment data. + +PostgreSQL infrastructure tables SHALL follow the same database backup, +encryption and access controls as the rest of CARE. + +--- + +## 60. Terraform Scope + +Terraform SHALL manage: + +- GCP APIs; +- Artifact Registry; +- service accounts; +- IAM; +- Cloud SQL; +- Cloud Storage; +- Secret Manager; +- Cloud Tasks; +- Cloud Scheduler; +- Cloud Run API; +- Cloud Run HTTP worker; +- Cloud Run Jobs; +- optional PostgreSQL queue worker service; +- monitoring; +- environment-specific configuration. + +PostgreSQL cache and queue schemas SHALL be created through migrations, +management commands or the queue library's schema tooling rather than Terraform +SQL embedded in infrastructure code. + +--- + +## 61. Deployment Pipeline + +The pipeline SHALL: + +1. run linting; +2. run tests; +3. build the image; +4. publish the image; +5. run migrations; +6. create or update cache and queue schemas; +7. deploy the selected worker type; +8. deploy the API; +9. update schedules; +10. run smoke tests; +11. record the revision. + +The pipeline SHALL detect the selected task and cache backends. + +--- + +## 62. Smoke Tests + +Smoke tests SHALL verify: + +```text +API health +database connectivity +selected cache backend +selected task backend +static files +authenticated API operation +storage write and read +worker execution +scheduled jobs +``` + +For PostgreSQL cache: + +```text +set +get +delete +expiration +``` + +For PostgreSQL queue: + +```text +enqueue +claim +execute +retry +complete +duplicate protection +``` + +--- + +## 63. Target Runtime Summary + +```text +CARE remains Django. + +Django ORM remains unchanged. + +PostgreSQL moves to Cloud SQL. + +PostgreSQL may also provide cache, rate limits, transient state, +task state and an optional task queue. + +Django Storage API handles all files through django-storages. + +All uploads and downloads pass through Django. + +Cloud Tasks remains the default queue for serverless GCP operation. + +A PostgreSQL-backed queue is available for consolidated deployments. + +Celery remains available for local and traditional deployments. + +Redis remains optional rather than mandatory. + +Cloud Scheduler replaces Celery Beat in the default GCP profile. + +Cloud Run Jobs handle migrations and batch work. + +No Compute Engine VM is required by the default GCP profile. +``` + +--- + +## 64. Definition of Done + +The target runtime is achieved when: + +- CARE runs on Cloud Run; +- PostgreSQL runs on Cloud SQL; +- Django Storage API handles all file operations; +- direct-to-bucket frontend flows are removed; +- Cloud Tasks works as the default GCP task backend; +- PostgreSQL is supported as an optional task backend; +- PostgreSQL is supported as a Django cache backend; +- PostgreSQL can support rate limits and transient shared state; +- Redis is optional; +- Upstash or another Redis-compatible provider can be selected; +- health checks follow selected backends; +- local Compose continues working; +- Celery remains supported; +- no permanent VM is required for the default profile; +- documentation clearly distinguishes the serverless and consolidated profiles. + +--- + +## 65. Next Document + +The next document is: + +```text +docs/xii/architecture/03-migration-plan.md +``` + +It will define how to implement: + +- Django Storage migration; +- Cloud Run and Cloud SQL deployment; +- PostgreSQL cache; +- optional PostgreSQL queue; +- Cloud Tasks; +- optional Redis; +- backend selection; +- frontend file-flow migration; +- testing and rollback; + +while keeping the fork deployable after every phase. + diff --git a/docs/xii/architecture/03-migration-plan.md b/docs/xii/architecture/03-migration-plan.md new file mode 100644 index 0000000000..01a6c0e457 --- /dev/null +++ b/docs/xii/architecture/03-migration-plan.md @@ -0,0 +1,2028 @@ +--- +title: GCP Implementation Plan +document: 03-migration-plan +version: 0.2.0 +status: Draft +source_repository: https://github.com/ohcnetwork/care +target_platform: Google Cloud Platform +deployment_type: Greenfield +depends_on: + - docs/xii/architecture/00-scope-and-goals.md + - docs/xii/architecture/01-current-runtime.md + - docs/xii/architecture/02-target-runtime.md +--- + +# GCP Implementation Plan + +## 1. Purpose + +This document defines the implementation sequence required to prepare CARE for +a new deployment on Google Cloud Platform. + +This is a greenfield deployment. + +There is no existing production installation to migrate. + +There are no existing production: + +- databases; +- users; +- patient records; +- uploaded files; +- object-storage buckets; +- Redis instances; +- Celery workers; +- virtual machines; +- scheduled jobs; +- production frontend integrations. + +The plan therefore focuses on adapting the CARE codebase and creating a new GCP +environment. + +It does not include: + +- production data migration; +- database replication; +- object copying; +- storage-provider cutover; +- synchronization between old and new systems; +- dual writes; +- compatibility windows for active users; +- rollback to an existing production installation. + +The implementation SHALL preserve compatibility with the official CARE +repository and its local development environment. + +--- + +## 2. Objectives + +The implementation SHALL produce a new CARE deployment using: + +- Cloud Run for the API; +- Cloud SQL for PostgreSQL; +- Cloud Storage through Django Storage API and `django-storages`; +- Cloud Tasks for the default GCP asynchronous task backend; +- a private Cloud Run service for Cloud Tasks execution; +- Cloud Scheduler for periodic triggers; +- Cloud Run Jobs for migrations, setup and batch work; +- PostgreSQL as an option for shared cache, rate limiting, transient state and + an optional task queue; +- Redis-compatible services as optional backends; +- Secret Manager for secrets; +- Artifact Registry for container images; +- Terraform for GCP infrastructure; +- an automated deployment pipeline. + +The default GCP deployment SHALL NOT require: + +- Compute Engine; +- a permanently running VM; +- self-managed MinIO; +- self-managed Redis; +- a permanent Celery worker; +- Celery Beat; +- direct frontend access to object storage; +- `boto3` for CARE file storage. + +--- + +## 3. Greenfield Assumptions + +The following assumptions apply throughout this plan. + +### 3.1 Empty production database + +The first production database will be created from CARE migrations. + +No previous schema or data must be imported. + +### 3.2 Empty production buckets + +Cloud Storage buckets will initially contain no CARE files. + +No MinIO, S3 or legacy object data must be copied. + +### 3.3 No legacy frontend deployment + +The production frontend can be deployed using the new API file flow from its +first release. + +There is no need to support signed-upload and server-mediated upload flows +simultaneously in production. + +### 3.4 No existing task infrastructure + +Cloud Tasks, Cloud Scheduler and Cloud Run Jobs can be introduced directly. + +There is no production Celery queue or Beat schedule to drain or disable. + +### 3.5 Local compatibility remains required + +The existing Docker Compose development model SHALL remain functional. + +Local CARE development MAY continue using: + +- PostgreSQL; +- Redis; +- Celery; +- Celery Beat; +- MinIO. + +Greenfield production does not mean local upstream behavior may be broken. + +--- + +## 4. Implementation Principles + +### 4.1 Build the desired runtime directly + +Production SHALL be created using the target architecture. + +The implementation SHALL not first deploy the legacy architecture and then +migrate it. + +### 4.2 Preserve upstream compatibility + +Changes to upstream-owned files SHALL remain: + +- small; +- focused; +- tested; +- easy to merge; +- easy to understand. + +Adding isolated settings, scripts and modules is preferred over modifying +shared files extensively. + +### 4.3 Use Django facilities first + +The implementation SHALL prefer established Django mechanisms. + +Examples: + +- Django ORM for persistence; +- Django Storage API for files; +- Django cache framework for cache selection; +- Django management commands for administrative and scheduled work; +- Django settings modules for deployment-specific configuration. + +### 4.4 Avoid unnecessary abstraction + +An abstraction SHALL be introduced only when multiple implementations are +actually supported. + +Examples: + +- Celery and Cloud Tasks; +- PostgreSQL cache and Redis cache; +- MinIO/S3 and GCS through Django Storage; +- optional PostgreSQL task queue. + +The implementation SHALL not introduce domain repositories or reorganize CARE +into new architectural layers. + +### 4.5 Keep every phase usable + +Each phase SHALL leave: + +- the repository buildable; +- tests runnable; +- local development functional; +- completed GCP components testable. + +--- + +## 5. Branch Strategy + +The recommended branches are: + +```text +upstream/develop + | + v +origin/develop + | + v +origin/gcp + | + v +feature/* +``` + +### 5.1 `origin/develop` + +`origin/develop` SHALL mirror `upstream/develop`. + +It SHALL not contain GCP-specific commits. + +### 5.2 `origin/gcp` + +`origin/gcp` SHALL contain the maintained GCP integration. + +It SHOULD remain deployable after each completed phase. + +### 5.3 Feature branches + +Each major phase SHOULD use a feature branch. + +Examples: + +```text +feature/gcp-settings +feature/gcp-container +feature/django-storages +feature/file-api +feature/cloud-tasks +feature/postgres-cache +feature/gcp-terraform +``` + +### 5.4 Upstream synchronization + +Official updates SHOULD be integrated through: + +```text +sync/upstream-YYYY-MM-DD +``` + +--- + +## 6. Commit Strategy + +Commits SHALL be small and focused. + +Recommended examples: + +```text +docs(gcp): document greenfield implementation plan +chore(gcp): add isolated GCP settings +chore(runtime): add production Cloud Run entrypoint +chore(storage): add django-storages dependencies +feat(storage): configure patient facility and report aliases +refactor(storage): save uploads through Django Storage API +refactor(storage): stream downloads through Django +feat(tasks): add task backend selection +feat(tasks): add Cloud Tasks dispatcher +feat(tasks): add authenticated Cloud Run worker endpoint +feat(cache): add PostgreSQL database cache +chore(gcp): add Terraform foundation +test(gcp): add Cloud Run smoke tests +``` + +Avoid large commits such as: + +```text +migrate CARE to GCP +complete cloud refactor +replace infrastructure +``` + +--- + +## 7. Phase Overview + +The greenfield implementation SHALL proceed through these phases: + +```text +Phase 0 Complete repository inventory +Phase 1 Establish test and branch baseline +Phase 2 Add isolated GCP settings +Phase 3 Build the production container +Phase 4 Create Terraform foundation +Phase 5 Deploy Cloud SQL and initialize CARE +Phase 6 Replace CARE file handling with django-storages +Phase 7 Implement server-mediated uploads and downloads +Phase 8 Implement configurable task execution +Phase 9 Add Cloud Tasks and the private worker +Phase 10 Add Cloud Scheduler and Cloud Run Jobs +Phase 11 Add PostgreSQL-backed cache and shared state +Phase 12 Add optional Redis-compatible backends +Phase 13 Evaluate the optional PostgreSQL task queue +Phase 14 Add health checks and observability +Phase 15 Add CI/CD and deployment automation +Phase 16 Verify the complete greenfield deployment +Phase 17 Validate upstream synchronization +``` + +No production-data migration phase is required. + +--- + +# Phase 0 — Complete Repository Inventory + +## 8. Objective + +Complete the technical inspection before changing CARE behavior. + +The existing `01-current-runtime.md` provides the initial runtime inventory, but +implementation requires a complete call-site inventory. + +## 9. Required searches + +Locate and document: + +- every `.delay()` call; +- every `.apply_async()` call; +- every `send_task()` call; +- every use of Celery result IDs; +- every use of the default Django cache; +- every `django_ratelimit` decorator; +- every direct Redis import; +- every use of `files_manager`; +- every signed-upload endpoint; +- every signed-download endpoint; +- every direct `boto3` or `botocore` import; +- every frontend upload call; +- every frontend download call; +- every periodic Celery registration; +- relevant plugin-provided behavior; +- production Dockerfiles and scripts; +- health-check routes; +- current CI workflows. + +## 10. Inventory documents + +Store the results under: + +```text +docs/xii/architecture/inventory/ +``` + +Recommended files: + +```text +storage-call-sites.md +task-call-sites.md +cache-and-redis.md +frontend-file-flow.md +runtime-and-deployment.md +plugin-impact.md +``` + +## 11. Exit criteria + +Phase 0 is complete when: + +- all known storage call sites are listed; +- all known task call sites are listed; +- all Redis and cache responsibilities are classified; +- frontend file flows are understood; +- unresolved plugin behavior is documented; +- no application behavior has changed. + +--- + +# Phase 1 — Test and Branch Baseline + +## 12. Objective + +Establish a reproducible starting point. + +## 13. Local baseline + +Run the current upstream-compatible workflow: + +```bash +make build +make up +make load-fixtures +make test +``` + +or the current official equivalents. + +Record: + +- test count; +- failures; +- skipped tests; +- lint result; +- migration status; +- container health; +- fixture-loading result. + +Existing failures SHALL be documented. + +## 14. Branch setup + +Configure: + +```text +origin +upstream +``` + +Example: + +```bash +git remote add upstream https://github.com/ohcnetwork/care.git +git fetch upstream +``` + +Ensure: + +```text +origin/develop +``` + +matches: + +```text +upstream/develop +``` + +Create or update: + +```text +origin/gcp +``` + +from the upstream mirror. + +## 15. Exit criteria + +Phase 1 is complete when: + +- the local stack starts; +- the baseline is documented; +- branch roles are established; +- `develop` mirrors upstream; +- `gcp` is ready for implementation. + +--- + +# Phase 2 — Isolated GCP Settings + +## 16. Objective + +Add a GCP deployment profile without rewriting existing settings. + +## 17. New module + +Create: + +```text +config/settings/gcp.py +``` + +It SHALL inherit from: + +```python +from .deployment import * +``` + +## 18. Initial responsibilities + +The module SHALL configure: + +- Cloud Run proxy behavior; +- secure host and origin handling; +- Cloud SQL connection values; +- task backend selection; +- cache backend selection; +- Redis optionality; +- backend-aware health checks; +- stdout and stderr logging; +- file-size limits; +- GCP storage aliases when Phase 6 is implemented. + +## 19. Required configuration + +Production SHALL require explicit values for sensitive settings such as: + +```text +DJANGO_SECRET_KEY +DATABASE_URL +CSRF_TRUSTED_ORIGINS +CORS_ALLOWED_ORIGINS +``` + +Unsafe development defaults SHALL not silently apply. + +## 20. Backend-selection variables + +The settings SHOULD recognize: + +```text +CARE_TASK_BACKEND=cloud_tasks|postgres|celery +CARE_CACHE_BACKEND=postgres|locmem|redis|dummy +CARE_RATE_LIMIT_BACKEND=postgres|redis +CARE_TRANSIENT_STATE_BACKEND=postgres|redis +``` + +Only implemented and tested combinations SHALL be accepted. + +Unsupported values SHALL cause a clear configuration error. + +## 21. Exit criteria + +Phase 2 is complete when: + +```bash +DJANGO_SETTINGS_MODULE=config.settings.deployment \ +python manage.py check +``` + +passes with a minimal valid environment. + +Local, test and deployment settings SHALL continue working. + +--- + +# Phase 3 — Production Container + +## 22. Objective + +Produce one immutable image usable by API, workers and jobs. + +## 23. Dockerfile + +Reuse an existing production Dockerfile when suitable. + +Otherwise add: + +```text +docker/gcp.Dockerfile +``` + +## 24. Image requirements + +The image SHALL: + +- install locked dependencies; +- install required GCP packages; +- install `django-storages` backends; +- compile translations; +- collect static files; +- use Gunicorn; +- avoid development tooling; +- run as non-root where practical; +- avoid embedded secrets; +- support multiple commands; +- avoid running migrations automatically. + +## 25. Runtime commands + +The image SHOULD support: + +```text +API +Cloud Tasks HTTP worker +optional PostgreSQL queue worker +Celery worker +Django management commands +``` + +Suggested scripts: + +```text +scripts/start-gcp-api.sh +scripts/start-gcp-task-worker.sh +scripts/start-postgres-queue-worker.sh +scripts/run-gcp-job.sh +``` + +## 26. API command + +The API SHALL listen on Cloud Run's `PORT`. + +Conceptually: + +```bash +gunicorn config.wsgi:application \ + --bind "0.0.0.0:${PORT}" \ + --workers "${GUNICORN_WORKERS}" \ + --threads "${GUNICORN_THREADS}" \ + --timeout "${GUNICORN_TIMEOUT}" +``` + +## 27. Exit criteria + +Phase 3 is complete when: + +- the production image builds; +- the API starts locally with Gunicorn; +- static files are present; +- the image executes management commands; +- the image does not require Redis unless selected; +- the development Dockerfile remains functional. + +--- + +# Phase 4 — Terraform Foundation + +## 28. Objective + +Create the GCP environment reproducibly before application deployment. + +## 29. Terraform layout + +Recommended structure: + +```text +deploy/gcp/terraform/ +├── modules/ +├── environments/ +│ ├── dev/ +│ ├── staging/ +│ └── prod/ +└── README.md +``` + +The initial implementation MAY begin with `dev` only. + +## 30. Initial resources + +Terraform SHALL create: + +- required GCP APIs; +- Artifact Registry; +- Cloud SQL; +- Cloud Storage buckets; +- Secret Manager resources; +- service accounts; +- IAM bindings; +- Cloud Run API service definition; +- Cloud Run task-worker definition; +- Cloud Run Job definitions; +- Cloud Tasks queues; +- Cloud Scheduler jobs; +- required networking; +- logging and monitoring foundations. + +Resources MAY be introduced incrementally as their application phase is +completed. + +## 31. Environments + +Because this is greenfield, environments can be created directly using the +target architecture. + +Suggested environments: + +```text +dev +staging +prod +``` + +Each SHOULD have separate: + +- database; +- buckets; +- queues; +- services; +- secrets; +- scheduled jobs. + +## 32. Exit criteria + +Phase 4 is complete when: + +- Terraform initializes; +- Terraform validates; +- a development plan can be reviewed; +- Artifact Registry exists; +- state storage is secured; +- IAM ownership is documented. + +--- + +# Phase 5 — Cloud SQL and Initial CARE Database + +## 33. Objective + +Create a new CARE database directly from Django migrations. + +## 34. Cloud SQL creation + +Create a development Cloud SQL PostgreSQL instance with: + +- a CARE database; +- a dedicated application user; +- backups; +- conservative sizing; +- controlled access; +- connection monitoring. + +No legacy database import is required. + +## 35. Migration job + +Create a Cloud Run Job using the same application image. + +The initial database SHALL be created by running: + +```bash +python manage.py migrate --noinput +``` + +## 36. Initial setup + +Run required setup commands explicitly: + +```bash +python manage.py sync_permissions_roles +python manage.py sync_valueset +``` + +Fixtures MAY be loaded in development or staging: + +```bash +python manage.py load_fixtures +``` + +Production fixture behavior SHALL be intentional and documented. + +## 37. Connection budget + +Calculate a conservative connection budget from: + +```text +API maximum instances +worker maximum instances +Gunicorn processes +Gunicorn threads +jobs +optional queue worker +``` + +Using PostgreSQL for cache or queues SHALL also be included in capacity +planning. + +## 38. Exit criteria + +Phase 5 is complete when: + +- Cloud SQL is created; +- migrations succeed on an empty database; +- setup commands succeed; +- CARE connects from Cloud Run; +- database health checks pass; +- no legacy data import is needed. + +--- + +# Phase 6 — Django Storage Implementation + +## 39. Objective + +Replace CARE's custom S3-specific file-management implementation with Django +Storage API before the first production deployment. + +Because the deployment is greenfield, no legacy storage compatibility period is +required in production. + +## 40. Dependencies + +Add `django-storages` with: + +```text +S3-compatible backend +Google Cloud Storage backend +``` + +Update the repository lockfile. + +## 41. Logical aliases + +Configure: + +```text +patient +facility +report +staticfiles +``` + +### Local configuration + +```text +patient -> S3Storage -> MinIO +facility -> S3Storage -> MinIO +report -> S3Storage -> MinIO +``` + +### GCP configuration + +```text +patient -> GoogleCloudStorage +facility -> GoogleCloudStorage +report -> GoogleCloudStorage +``` + +The `report` alias MAY use the same physical bucket as `patient` while +remaining logically separate. + +## 42. Refactor storage operations + +Replace CARE file operations with: + +```python +from django.core.files.storage import storages +``` + +Use Django Storage methods such as: + +```text +save +open +exists +delete +size +``` + +Remove storage-specific URL construction from application logic. + +## 43. Existing object names + +The current naming convention MAY be retained: + +```text +/ +``` + +No object migration is required because production buckets are empty. + +## 44. Compatibility wrapper + +A thin wrapper MAY temporarily preserve existing internal method signatures +during code refactoring. + +It SHALL delegate only to Django Storage API. + +It SHALL NOT: + +- create `boto3` clients; +- create GCS clients; +- generate signed URLs; +- duplicate provider CRUD logic. + +The wrapper SHOULD be removed if direct Django Storage use produces a cleaner +and sufficiently small upstream patch. + +## 45. Storage tests + +Test both: + +```text +MinIO through S3Storage +GCS through GoogleCloudStorage +``` + +Required operations: + +- save; +- open; +- streaming read; +- exists; +- delete; +- Unicode names; +- duplicate names; +- content type; +- large file within supported limits; +- missing object; +- permission errors. + +## 46. Exit criteria + +Phase 6 is complete when: + +- local storage works through MinIO and `django-storages`; +- GCP storage works through GCS and `django-storages`; +- CARE storage code no longer calls `boto3`; +- production buckets remain empty until the application is launched; +- no storage-data migration is required. + +--- + +# Phase 7 — Server-Mediated File API + +## 47. Objective + +Implement the only production file flow before the frontend is deployed. + +All uploads and downloads SHALL pass through CARE. + +There is no need for a production compatibility window with signed URLs. + +## 48. Upload endpoints + +The API SHALL: + +1. authenticate the caller; +2. authorize the operation; +3. validate metadata; +4. validate extension; +5. validate MIME type; +6. enforce size limits; +7. save through the correct storage alias; +8. create or update the CARE record; +9. return provider-neutral metadata. + +## 49. Upload handling + +The implementation SHOULD pass Django uploaded-file objects directly to +storage. + +It SHALL avoid reading entire files into memory unnecessarily. + +Configure: + +```text +DATA_UPLOAD_MAX_MEMORY_SIZE +FILE_UPLOAD_MAX_MEMORY_SIZE +``` + +and a documented maximum upload size. + +## 50. Download endpoints + +The API SHALL: + +1. authenticate the caller; +2. authorize access; +3. open the object through Django Storage; +4. return `FileResponse` or equivalent streaming response; +5. set safe content type; +6. set safe content disposition. + +The API SHALL not return normal direct bucket URLs. + +## 51. Frontend implementation + +The first production frontend SHALL use the new API flow. + +It SHALL NOT implement or retain production support for: + +- signed PUT URLs; +- direct MinIO uploads; +- direct S3 uploads; +- direct GCS uploads; +- signed storage downloads; +- external bucket endpoints. + +## 52. Security tests + +Test: + +- unauthenticated access; +- unauthorized access; +- patient and facility boundaries; +- filename handling; +- MIME validation; +- extension validation; +- file-size limits; +- streaming behavior; +- missing objects; +- deleted records; +- storage failures. + +## 53. Exit criteria + +Phase 7 is complete when: + +- the backend file API works; +- the frontend uses it; +- no production direct-to-bucket flow exists; +- memory usage is acceptable; +- Cloud Run request duration is acceptable; +- no legacy production frontend must be supported. + +--- + +# Phase 8 — Task Inventory and Reusable Task Logic + +## 54. Objective + +Prepare current Celery tasks for configurable execution. + +## 55. Task analysis + +For each task, document: + +- call sites; +- arguments; +- result usage; +- duration; +- retries; +- expiry; +- idempotency; +- storage access; +- database effects; +- email effects; +- periodic scheduling; +- plugin ownership. + +## 56. Task classification + +Classify each task as: + +```text +Cloud Tasks +Cloud Run Job +synchronous +Celery compatibility +requires further analysis +``` + +## 57. Reusable functions + +Extract task behavior from Celery decorators only where needed. + +Target structure: + +```text +reusable implementation +thin Celery wrapper +thin Cloud Tasks handler +optional PostgreSQL queue wrapper +``` + +Preserve existing task names where practical. + +## 58. Dispatch API + +Add a narrow API conceptually similar to: + +```python +enqueue_task( + task_name, + payload, + delay_seconds=None, + task_id=None, +) +``` + +The implementation SHALL not include unused Celery workflow concepts. + +## 59. Transaction timing + +Use: + +```python +transaction.on_commit(...) +``` + +where task dispatch must occur only after database commit. + +## 60. Exit criteria + +Phase 8 is complete when: + +- all core tasks are classified; +- reusable logic exists where required; +- Celery still works locally; +- task payloads are JSON-serializable; +- callers do not depend on undocumented Celery behavior. + +--- + +# Phase 9 — Cloud Tasks and Private Worker + +## 61. Objective + +Implement the default GCP asynchronous execution path. + +## 62. Cloud Tasks dispatcher + +The dispatcher SHALL: + +- use the official Google client; +- use Application Default Credentials; +- enqueue HTTP tasks; +- serialize JSON; +- support delays where required; +- attach OIDC identity; +- return the task name; +- avoid logging sensitive payloads. + +## 63. Private worker + +Deploy a private Cloud Run service with: + +```text +minimum instances: 0 +allow unauthenticated: false +``` + +The worker SHALL: + +- accept only POST; +- validate payloads; +- allow only registered tasks; +- reject arbitrary callables; +- log task metadata; +- return non-2xx on retriable failure; +- remain idempotent. + +## 64. IAM + +Use a dedicated task-invoker service account. + +It SHALL receive permission to invoke only the worker service. + +## 65. Pilot task + +Select one bounded, low-risk task that: + +- does not require Celery result retrieval; +- has a small payload; +- is easy to observe; +- can tolerate retries; +- can be made idempotent. + +Test: + +- enqueue; +- scale from zero; +- execution; +- retry; +- duplicate delivery; +- authorization; +- failure logging. + +## 66. Remaining tasks + +Migrate request-triggered tasks individually. + +The default expected mapping is: + +| CARE task | GCP execution | +|---|---| +| TOTP enabled email | Cloud Tasks | +| TOTP disabled email | Cloud Tasks | +| report generation | Cloud Tasks, after duration testing | + +## 67. Exit criteria + +Phase 9 is complete when: + +- Cloud Tasks enqueues successfully; +- the worker scales from zero; +- IAM blocks unauthorized callers; +- migrated tasks execute reliably; +- local Celery execution still passes tests. + +--- + +# Phase 10 — Cloud Scheduler and Cloud Run Jobs + +## 68. Objective + +Implement periodic and administrative work directly in the target GCP model. + +There is no production Celery Beat process to transition or drain. + +## 69. Management commands + +Expose reusable periodic logic as Django management commands. + +Expected commands include: + +```text +cleanup_expired_token_slots +cleanup_incomplete_file_uploads +``` + +## 70. Jobs + +Create Cloud Run Jobs for: + +```text +migrate +sync_permissions_roles +sync_valueset +load_fixtures, in approved environments +cleanup_expired_token_slots +cleanup_incomplete_file_uploads +other batch operations +``` + +## 71. Scheduler + +Create Cloud Scheduler triggers for periodic jobs. + +Reproduce the intended CARE schedules: + +```text +expired token cleanup: daily +incomplete upload cleanup: according to configured expiry +``` + +## 72. Idempotency + +Jobs SHALL tolerate retries and repeated invocation. + +Cleanup jobs SHALL process empty databases and empty buckets successfully. + +This is especially important for the initial greenfield deployment, where +scheduled jobs may run before substantial data exists. + +## 73. Exit criteria + +Phase 10 is complete when: + +- migrations run as jobs; +- setup commands run as jobs; +- periodic cleanup runs through Scheduler and Jobs; +- no production Celery Beat is deployed; +- local Celery Beat remains available. + +--- + +# Phase 11 — PostgreSQL Cache and Shared State + +## 74. Objective + +Support a Redis-free GCP profile using the already required PostgreSQL service. + +## 75. Database cache + +Add support for: + +```text +django.core.cache.backends.db.DatabaseCache +``` + +Example configuration: + +```python +CACHES = { + "default": { + "BACKEND": "django.core.cache.backends.db.DatabaseCache", + "LOCATION": "care_cache", + }, +} +``` + +Create the table during initial environment setup: + +```bash +python manage.py createcachetable +``` + +Because the database is new, no cache migration is required. + +## 76. Backend selection + +Support: + +```text +CARE_CACHE_BACKEND=postgres +CARE_CACHE_BACKEND=locmem +CARE_CACHE_BACKEND=redis +CARE_CACHE_BACKEND=dummy +``` + +## 77. Appropriate PostgreSQL uses + +PostgreSQL MAY support: + +- shared Django cache; +- report progress; +- shared transient state; +- rate-limit counters; +- task execution state; +- idempotency. + +Use explicit models instead of cache tables when state must be: + +- durable; +- auditable; +- queryable; +- correctness-critical. + +## 78. Report progress + +Choose between: + +```text +Django database cache +explicit PostgreSQL model +``` + +A model SHOULD be used when task failures or historical progress must remain +visible. + +## 79. Rate limiting + +Test the current rate-limiting library with the PostgreSQL-backed cache. + +If required atomicity is not guaranteed, implement explicit PostgreSQL +counters using transactions and constraints. + +## 80. Capacity testing + +Measure: + +- cache queries per request; +- rate-limit writes; +- task-progress writes; +- table growth; +- expired-row cleanup; +- database latency; +- connection usage. + +## 81. Exit criteria + +Phase 11 is complete when: + +- the API starts without Redis; +- the worker starts without Redis; +- shared cache works across instances; +- report progress works across instances; +- rate limiting remains correct; +- PostgreSQL load is acceptable. + +--- + +# Phase 12 — Optional Redis-Compatible Backends + +## 82. Objective + +Allow Redis-compatible services when they provide a measurable benefit. + +Redis SHALL remain optional. + +## 83. Supported responsibilities + +Redis MAY provide: + +- cache; +- rate limiting; +- shared transient state; +- Celery broker in traditional deployments; +- Celery result backend. + +## 84. Independent configuration + +Support variables such as: + +```text +REDIS_CACHE_URL +REDIS_RATE_LIMIT_URL +REDIS_TRANSIENT_STATE_URL +``` + +The variables MAY point to one instance but SHALL remain logically separate. + +## 85. Provider neutrality + +Use standard Redis-compatible clients and URLs. + +An Upstash deployment SHOULD use: + +```text +CARE_CACHE_BACKEND=redis +REDIS_CACHE_URL=rediss://... +``` + +The application SHOULD not require an Upstash-specific switch. + +## 86. Failure behavior + +Document behavior for each responsibility. + +Examples: + +```text +performance cache failure -> cache miss +rate-limit failure -> defined security fallback +progress failure -> controlled degradation +Celery broker failure -> dispatch error +``` + +## 87. Exit criteria + +Phase 12 is complete when: + +- Redis can be enabled selectively; +- the Redis-free profile still works; +- TLS configuration is tested; +- Upstash-compatible configuration is documented; +- Redis checks run only when required. + +--- + +# Phase 13 — Optional PostgreSQL Task Queue Evaluation + +## 88. Objective + +Determine whether CARE should support a PostgreSQL-backed task queue in addition +to Cloud Tasks and Celery. + +This phase SHALL NOT block the default GCP deployment. + +## 89. Evaluation scope + +Evaluate a maintained queue system intentionally designed for PostgreSQL. + +Do not use an obsolete Celery SQL broker transport. + +Required capabilities: + +- Django compatibility; +- PostgreSQL-native locking; +- retries; +- scheduled execution; +- concurrency controls; +- task state; +- transactional enqueue; +- schema management; +- cleanup; +- observability. + +## 90. Greenfield advantage + +Because there is no production queue, the candidate can be tested on an empty +schema without converting Celery messages or preserving queued jobs. + +No queue migration is required. + +## 91. Worker implications + +Document that an immediate PostgreSQL queue needs an active consumer. + +Possible modes: + +```text +Cloud Run service with minimum instance 1 +traditional worker container +Kubernetes or on-premise worker +periodic Cloud Run Job for non-urgent work +``` + +A PostgreSQL queue SHALL not be described as equivalent to Cloud Tasks for +scale-to-zero behavior. + +## 92. Benchmark + +Measure: + +- enqueue latency; +- claim latency; +- polling or notification behavior; +- database connections; +- table growth; +- retries; +- throughput; +- effect on CARE queries; +- minimum worker cost. + +## 93. Decision + +Produce an ADR with one of: + +```text +accepted +accepted for limited deployment profiles +deferred +rejected +``` + +## 94. Exit criteria + +Phase 13 is complete when the project has a documented decision. + +The default Cloud Tasks deployment may be completed regardless of the outcome. + +--- + +# Phase 14 — Health Checks and Observability + +## 95. Objective + +Make operational behavior reflect the selected runtime profile. + +## 96. Health categories + +Define: + +```text +liveness +readiness +dependency diagnostics +``` + +The mandatory GCP readiness checks SHALL cover: + +- application startup; +- PostgreSQL. + +Additional checks SHALL depend on configuration. + +Examples: + +### Cloud Tasks profile + +```text +no Celery queue check +no Redis check unless Redis is enabled +``` + +### PostgreSQL cache profile + +```text +database and cache table availability +``` + +### Redis profile + +```text +Redis only when required by selected responsibilities +``` + +### PostgreSQL queue profile + +```text +queue schema and worker freshness +``` + +## 97. Structured logging + +Include: + +```text +service +revision +environment +request ID +task ID +task backend +task name +attempt +duration +status +``` + +## 98. Sensitive-data review + +Logs SHALL not include: + +- passwords; +- tokens; +- complete patient records; +- full task payloads; +- file contents; +- storage credentials; +- secret values. + +## 99. Initial alerts + +Configure alerts for: + +- API error rate; +- task failures; +- scheduled-job failures; +- Cloud SQL connections; +- Cloud SQL storage; +- storage-access failures; +- task backlog; +- optional Redis failures. + +## 100. Exit criteria + +Phase 14 is complete when: + +- health checks match the active profile; +- logs are structured; +- sensitive logging is reviewed; +- essential alerts exist. + +--- + +# Phase 15 — CI/CD and Deployment Automation + +## 101. Objective + +Deploy immutable revisions into an empty or newly initialized environment. + +## 102. Pipeline sequence + +The pipeline SHALL: + +1. run formatting checks; +2. run linting; +3. run tests; +4. build the image; +5. tag the image with the commit SHA; +6. push to Artifact Registry; +7. create or update job definitions; +8. run database migrations; +9. create the database-cache table when selected; +10. run required setup commands; +11. deploy the task worker; +12. deploy the API; +13. create or update scheduled jobs; +14. run smoke tests; +15. record the deployed revision. + +## 103. First deployment + +The first deployment has no previous production revision. + +It SHALL: + +- create the empty infrastructure; +- initialize the database; +- initialize cache or queue schemas; +- deploy services; +- verify the complete stack. + +## 104. Later application rollback + +After the first deployment, application rollback MAY deploy a previous +immutable Cloud Run revision. + +Database-schema compatibility SHALL be checked. + +There is no rollback to a legacy CARE infrastructure because none exists. + +## 105. Exit criteria + +Phase 15 is complete when: + +- a clean environment can be deployed automatically; +- migrations are explicit; +- services use immutable images; +- smoke tests gate deployment success; +- subsequent revision rollback is documented. + +--- + +# Phase 16 — Complete Greenfield Verification + +## 106. Objective + +Verify the complete new installation before real use. + +## 107. Functional tests + +Verify: + +- initial administrative access; +- authentication; +- permissions; +- facilities; +- patients; +- encounters; +- file upload; +- file download; +- report generation; +- email tasks; +- periodic cleanup; +- cache; +- rate limiting; +- audit logs; +- relevant plugins. + +Use only synthetic test data. + +## 108. Scaling tests + +Verify: + +- API scale from zero; +- worker scale from zero; +- concurrent requests; +- database connection limits; +- storage streaming; +- task retries; +- duplicate task delivery; +- scheduled-job execution. + +## 109. Empty-state tests + +Because the installation is new, explicitly test: + +- empty database screens and APIs; +- empty buckets; +- cleanup jobs with no records; +- report endpoints before templates exist; +- first-user and first-facility workflows; +- fixture and value-set initialization. + +## 110. Failure tests + +Simulate: + +- Cloud SQL interruption; +- GCS permission failure; +- task exception; +- duplicate task request; +- email-provider failure; +- optional Redis outage; +- failed migration in a disposable environment. + +## 111. Cost validation + +Measure expected cost for: + +```text +Cloud SQL +Cloud Run +Cloud Tasks +Cloud Storage +Cloud Scheduler +Artifact Registry +Secret Manager +Cloud Logging +optional Redis +optional active PostgreSQL queue worker +``` + +## 112. Exit criteria + +Phase 16 is complete when: + +- functional tests pass; +- empty-state behavior works; +- connection usage is safe; +- storage streaming is acceptable; +- task behavior is reliable; +- expected costs are documented; +- the environment is ready for first real use. + +--- + +# Phase 17 — Upstream Synchronization Validation + +## 113. Objective + +Prove that the GCP fork remains maintainable. + +## 114. Synchronization process + +Use: + +```bash +git fetch upstream + +git switch develop +git reset --hard upstream/develop +git push --force-with-lease origin develop + +git switch gcp +git switch -c sync/upstream-YYYY-MM-DD +git merge develop +``` + +## 115. Validation + +After conflict resolution, run: + +```bash +make build +make up +make test +``` + +Also verify: + +```text +GCP settings +production image +Django storage aliases +file API +Cloud Tasks dispatch +PostgreSQL cache +Terraform validation +``` + +## 116. Repeated conflicts + +If the same upstream file repeatedly conflicts, move GCP behavior into: + +- a separate settings module; +- a helper; +- a new script; +- configuration; +- a narrowly scoped extension point. + +## 117. Exit criteria + +Phase 17 is complete when: + +- a current upstream merge succeeds; +- local tests pass; +- GCP tests pass; +- recurring conflicts are documented; +- the synchronization procedure is reproducible. + +--- + +## 118. Test Matrix + +The implementation SHALL test these profiles: + +| Profile | Database | Storage | Tasks | Cache | +|---|---|---|---|---| +| Local upstream-compatible | PostgreSQL container | MinIO through `S3Storage` | Celery + Redis | Redis | +| GCP default | Cloud SQL | GCS through `GoogleCloudStorage` | Cloud Tasks | PostgreSQL | +| GCP optional Redis | Cloud SQL | GCS | Cloud Tasks | Redis-compatible | +| Consolidated PostgreSQL | PostgreSQL | configured Django storage | PostgreSQL queue | PostgreSQL | +| Test | test database | temporary test storage | fake/eager backend | Dummy or LocMem | + +The consolidated PostgreSQL profile is required only if Phase 13 accepts it. + +--- + +## 119. Simplified Rollback Model + +Because the deployment is greenfield, rollback is limited to newly deployed +code and infrastructure. + +### Before first production use + +The entire environment MAY be recreated from Terraform and initialized again. + +No production data preservation is required before real use begins. + +### After real use begins + +Normal operational protections become necessary: + +- Cloud SQL backups; +- point-in-time recovery; +- Cloud Storage retention; +- immutable application revisions; +- migration compatibility. + +### Application rollback + +Deploy a previous immutable Cloud Run revision. + +### Cache backend changes + +Cache values are disposable. + +Changing among: + +```text +postgres +redis +locmem +dummy +``` + +does not require data migration. + +### Task backend changes + +Switching task backends requires the selected worker or service to exist. + +Because no legacy queue is migrated during initial deployment, there are no +old queued messages to preserve. + +### Infrastructure recreation + +Development and staging environments SHOULD be reproducible from Terraform. + +Production stateful resources SHALL not be destroyed casually after real data +exists. + +--- + +## 120. Implementation Risks + +### 120.1 File traffic through Cloud Run + +Files pass through Django. + +Potential effects: + +- request duration; +- memory usage; +- Cloud Run bandwidth; +- maximum file size. + +Mitigations: + +- streaming; +- Django temporary-file handlers; +- explicit limits; +- adequate CPU and memory; +- realistic tests before first use. + +### 120.2 Cloud SQL load + +PostgreSQL may support: + +- CARE data; +- cache; +- rate limiting; +- progress; +- optional queueing. + +The implementation SHALL measure whether consolidation requires a larger +instance. + +### 120.3 Task duplication + +Cloud Tasks and PostgreSQL queues may retry. + +Task handlers SHALL be idempotent before production use. + +### 120.4 Frontend file-flow changes + +The frontend must implement server-mediated files from its first production +release. + +There is no need for a legacy production compatibility flow. + +### 120.5 Upstream changes + +Storage, settings and task files may change upstream. + +Custom patches SHALL remain narrow. + +### 120.6 Plugins + +Plugins may assume: + +- Celery; +- Redis; +- S3 APIs; +- signed URLs. + +Required plugins SHALL be tested before first production deployment. + +--- + +## 121. Deliverables + +The complete implementation SHALL produce: + +```text +config/settings/gcp.py +production container definition +Cloud Run API entrypoint +Cloud Run task-worker entrypoint +Cloud Run Job definitions +Django Storage aliases +server-mediated upload API +server-mediated download API +task dispatcher +Cloud Tasks backend +PostgreSQL cache support +optional Redis support +optional PostgreSQL queue decision +Terraform +deployment pipeline +test suite +operations documentation +upstream synchronization documentation +``` + +--- + +## 122. Definition of Completion + +The implementation is complete when: + +- CARE can be deployed from scratch into a new GCP project; +- no Compute Engine VM is required; +- the API runs on Cloud Run; +- the Cloud Tasks worker runs privately on Cloud Run; +- both services can scale to zero; +- Cloud SQL is initialized directly from Django migrations; +- no production data migration is required; +- Cloud Storage buckets are created empty; +- all file traffic passes through Django; +- `django-storages` handles MinIO and GCS; +- no CARE storage code uses `boto3`; +- Cloud Tasks handles default asynchronous work; +- Scheduler and Jobs handle periodic and administrative work; +- PostgreSQL can provide shared cache and state; +- Redis remains optional; +- the optional PostgreSQL queue has an explicit decision; +- local Docker Compose remains functional; +- the frontend uses the new file API from its first production release; +- the environment passes greenfield acceptance tests; +- upstream synchronization is proven. + +--- + +## 123. Next Document + +The next document is: + +```text +docs/xii/architecture/04-testing.md +``` + +It will define: + +- unit tests; +- storage integration tests; +- file API tests; +- task-backend tests; +- PostgreSQL cache tests; +- optional PostgreSQL queue tests; +- Redis compatibility tests; +- Cloud Run smoke tests; +- empty-state tests; +- security tests; +- upstream compatibility gates. diff --git a/docs/xii/architecture/04-testing.md b/docs/xii/architecture/04-testing.md new file mode 100644 index 0000000000..9957c18adb --- /dev/null +++ b/docs/xii/architecture/04-testing.md @@ -0,0 +1,1992 @@ +--- +title: GCP Testing Strategy +document: 04-testing +version: 0.1.0 +status: Draft +source_repository: https://github.com/ohcnetwork/care +target_platform: Google Cloud Platform +deployment_type: Greenfield +depends_on: + - docs/xii/architecture/00-scope-and-goals.md + - docs/xii/architecture/01-current-runtime.md + - docs/xii/architecture/02-target-runtime.md + - docs/xii/architecture/03-migration-plan.md +--- + +# GCP Testing Strategy + +## 1. Purpose + +This document defines the testing strategy for the greenfield GCP adaptation of +CARE. + +The test suite SHALL verify that: + +- existing CARE behavior remains functional; +- local Docker Compose development remains supported; +- GCP settings are valid; +- Cloud Run services start correctly; +- Cloud SQL works with Django ORM; +- all file operations use Django Storage API; +- MinIO and GCS behave consistently for CARE use cases; +- uploads and downloads pass through Django; +- Cloud Tasks executes registered tasks securely; +- PostgreSQL can provide shared cache and state; +- Redis remains optional; +- optional Redis-compatible deployments work when enabled; +- an optional PostgreSQL task queue can be evaluated without weakening the + default Cloud Tasks profile; +- upstream updates remain testable. + +The strategy is designed for a new deployment. + +It does not include: + +- production-data migration tests; +- legacy-object copy verification; +- dual-write tests; +- coexistence with an existing production installation; +- cutover from a running legacy stack. + +--- + +## 2. Testing Principles + +### 2.1 Preserve upstream tests + +The official CARE test suite SHALL remain the primary regression baseline. + +GCP-specific tests SHALL extend the existing suite rather than replace it. + +### 2.2 Test behavior, not implementation details + +Tests SHOULD verify observable behavior. + +Examples: + +```text +file can be uploaded and downloaded +task is enqueued and executed +cache value is shared across processes +unauthorized worker invocation is rejected +``` + +Tests SHOULD avoid asserting unnecessary internal details such as: + +- exact private method names; +- internal client construction; +- provider SDK object shapes; +- Terraform resource ordering. + +### 2.3 Test supported deployment profiles + +At minimum, tests SHALL cover: + +```text +local upstream-compatible profile +GCP default profile +GCP PostgreSQL-cache profile +GCP optional Redis profile +``` + +The PostgreSQL queue profile SHALL be tested only if that backend is accepted. + +### 2.4 Keep tests deterministic + +Tests SHALL avoid: + +- real patient data; +- uncontrolled external dependencies; +- arbitrary sleeps; +- dependence on previous test order; +- persistent shared state between runs; +- manually prepared cloud resources. + +### 2.5 Separate test levels + +The suite SHALL distinguish: + +```text +unit tests +application integration tests +provider integration tests +container tests +infrastructure validation +deployment smoke tests +end-to-end tests +``` + +A failure should identify the responsible layer. + +--- + +## 3. Test Categories + +The project SHALL organize tests into the following categories. + +```text +tests/ +├── unit/ +├── integration/ +│ ├── storage/ +│ ├── tasks/ +│ ├── cache/ +│ ├── database/ +│ └── redis/ +├── api/ +│ └── files/ +├── container/ +├── gcp/ +│ ├── cloud_run/ +│ ├── cloud_tasks/ +│ ├── cloud_sql/ +│ └── cloud_storage/ +├── security/ +├── smoke/ +└── upstream/ +``` + +This structure is illustrative. + +The implementation MAY follow existing CARE test conventions instead of +creating this exact directory layout. + +The conceptual separation SHALL remain. + +--- + +# 4. Upstream Regression Tests + +## 4.1 Objective + +Ensure GCP changes do not break ordinary CARE development. + +## 4.2 Required commands + +The official local workflow SHALL continue to run: + +```bash +make build +make up +make load-fixtures +make test +``` + +or the current official equivalents. + +The regression gate SHALL verify: + +- PostgreSQL container starts; +- Redis container starts; +- MinIO container starts; +- backend container starts; +- Celery container starts; +- migrations succeed; +- fixtures load; +- official tests pass; +- no GCP credentials are required. + +## 4.3 Local task verification + +The local profile SHALL continue testing: + +```text +Celery +Redis broker +Redis result backend +Celery Beat +``` + +GCP-specific changes SHALL not silently change local task behavior. + +## 4.4 Local storage verification + +Local tests SHALL verify that MinIO works through: + +```text +django-storages +S3Storage +``` + +The tests SHALL no longer depend on CARE's direct `boto3` file manager after +the storage migration is complete. + +## 4.5 Exit gate + +No GCP feature branch SHALL merge if it breaks the supported local profile. + +--- + +# 5. Settings Tests + +## 5.1 Objective + +Verify that settings modules construct valid configurations for each supported +profile. + +## 5.2 GCP settings import + +Test: + +```bash +DJANGO_SETTINGS_MODULE=config.settings.deployment \ +python manage.py check +``` + +with valid environment variables. + +## 5.3 Missing configuration + +Tests SHALL verify clear failures when required variables are absent. + +Examples: + +```text +DJANGO_SECRET_KEY +DATABASE_URL +GCP_PROJECT_ID +CARE_PATIENT_STORAGE_BUCKET +CARE_FACILITY_STORAGE_BUCKET +CARE_REPORT_STORAGE_BUCKET +GCP_TASKS_QUEUE +GCP_WORKER_URL +``` + +Only variables required by the selected backend SHALL be mandatory. + +For example: + +```text +REDIS_CACHE_URL +``` + +SHALL not be required when: + +```text +CARE_CACHE_BACKEND=postgres +``` + +## 5.4 Invalid backend names + +Tests SHALL reject values such as: + +```text +CARE_TASK_BACKEND=unknown +CARE_CACHE_BACKEND=memcached-guess +CARE_RATE_LIMIT_BACKEND=invalid +``` + +The resulting error SHALL identify: + +- the invalid variable; +- the invalid value; +- supported values. + +## 5.5 Profile combinations + +Test at least: + +```text +cloud_tasks + postgres cache +cloud_tasks + redis cache +celery + redis cache +cloud_tasks + locmem cache +``` + +If accepted: + +```text +postgres queue + postgres cache +``` + +## 5.6 Health-check construction + +Verify that: + +- Celery checks exist only when Celery is selected; +- Redis checks exist only when Redis is required; +- PostgreSQL cache checks exist when database cache is selected; +- GCP settings do not require Redis in the default profile. + +--- + +# 6. Database Tests + +## 6.1 Objective + +Verify CARE works correctly with PostgreSQL and GCP-specific connection +settings. + +## 6.2 Django ORM regression + +The GCP adaptation SHALL reuse existing CARE ORM tests. + +No separate persistence abstraction is required. + +## 6.3 Migration tests + +On an empty database, run: + +```bash +python manage.py migrate --noinput +``` + +The command SHALL succeed from a clean schema. + +The test SHALL detect: + +- missing migrations; +- migration ordering issues; +- plugin migration failures; +- assumptions about pre-existing tables. + +## 6.4 Setup commands + +Verify from an empty database: + +```bash +python manage.py sync_permissions_roles +python manage.py sync_valueset +``` + +Development tests MAY also run: + +```bash +python manage.py load_fixtures +``` + +## 6.5 Cache-table creation + +When PostgreSQL cache is selected, verify: + +```bash +python manage.py createcachetable +``` + +creates the configured table. + +The command SHOULD be idempotent or handled safely in deployment automation. + +## 6.6 Connection tests + +Container and GCP integration tests SHALL verify: + +- initial connection; +- connection reuse; +- stale connection recovery; +- concurrent requests; +- maximum-instance connection budget; +- API and worker simultaneous access. + +## 6.7 Transaction tests + +Task dispatch tests SHALL verify: + +```python +transaction.on_commit(...) +``` + +behavior where task creation depends on committed state. + +A task SHALL not execute against data that was rolled back. + +--- + +# 7. Django Storage Tests + +## 7.1 Objective + +Verify CARE uses Django Storage API consistently across MinIO and GCS. + +## 7.2 Storage test matrix + +The same logical test cases SHALL run against: + +```text +S3Storage connected to MinIO +GoogleCloudStorage connected to GCS +``` + +A lightweight temporary filesystem backend MAY be used for fast unit tests, but +it SHALL not replace provider integration tests. + +## 7.3 Alias tests + +Verify that these aliases resolve: + +```text +patient +facility +report +staticfiles +``` + +Each alias SHALL point to the expected backend in each environment. + +## 7.4 Basic operations + +Test: + +```text +save +open +exists +delete +size +name normalization +streaming read +``` + +## 7.5 Object naming + +Verify the expected naming convention: + +```text +/ +``` + +Test: + +- nested prefixes; +- Unicode; +- whitespace; +- long names; +- repeated names; +- extensions; +- generated internal names. + +## 7.6 Duplicate names + +Django Storage may modify a name when an object already exists. + +Tests SHALL verify CARE's intended behavior for duplicate names. + +If CARE requires stable object names, the implementation SHALL explicitly +delete, overwrite or generate unique names according to CARE semantics. + +Tests SHALL not assume identical overwrite behavior across providers without +verification. + +## 7.7 Content types + +Verify that saved files preserve or expose the expected content type where +supported. + +CARE responses SHALL not trust storage metadata without validation. + +## 7.8 Missing objects + +Verify behavior when: + +- database record exists but object does not; +- object is deleted before download; +- storage raises a not-found error. + +The API SHALL return a controlled CARE-level response rather than an unhandled +provider exception. + +## 7.9 Permission failures + +Simulate or test: + +- read denied; +- write denied; +- delete denied; +- bucket missing. + +Provider exceptions SHALL be handled consistently. + +## 7.10 Provider isolation + +Tests SHALL ensure CARE file code does not import: + +```text +boto3 +google.cloud.storage +``` + +outside approved dependency modules. + +A static-analysis or import-boundary test MAY enforce this. + +--- + +# 8. File Upload API Tests + +## 8.1 Objective + +Verify that all uploads pass through authenticated CARE endpoints. + +## 8.2 Successful upload + +Test: + +1. authenticated request; +2. authorized user; +3. valid file; +4. storage save; +5. database record creation or update; +6. provider-neutral API response. + +## 8.3 Authentication + +Verify unauthenticated requests are rejected. + +## 8.4 Authorization + +Verify users cannot upload files for: + +- unauthorized patients; +- unauthorized facilities; +- inaccessible encounters; +- other tenants or organizations. + +Tests SHALL follow CARE's actual authorization model. + +## 8.5 Extension validation + +Test: + +- allowed extensions; +- blocked executable extensions; +- missing extension; +- double extension; +- uppercase extension; +- misleading extension. + +## 8.6 MIME validation + +Test: + +- allowed MIME type; +- mismatched MIME type; +- missing MIME type; +- unsafe MIME type; +- browser-provided incorrect MIME type. + +The implementation SHOULD validate using more than the filename when current +CARE policy requires it. + +## 8.7 File-size limits + +Test: + +- file below memory threshold; +- file above memory threshold but below maximum; +- file exactly at maximum; +- file above maximum; +- request above total upload limit. + +## 8.8 Temporary upload handling + +For files above the in-memory threshold, verify Django uses temporary-file +handling rather than loading the complete file into memory. + +## 8.9 Storage failure + +Simulate storage failure after request validation. + +Verify: + +- no false success response; +- database consistency; +- no completed record referencing a missing object; +- appropriate logging. + +## 8.10 Database failure + +Simulate database failure after storage write. + +Verify the implementation performs its documented cleanup or leaves a +detectable incomplete object for cleanup. + +## 8.11 Empty file + +Test whether zero-byte files are: + +- accepted; +- rejected; +- accepted only for selected types. + +The behavior SHALL be explicit. + +--- + +# 9. File Download API Tests + +## 9.1 Objective + +Verify that downloads pass through CARE authorization and streaming. + +## 9.2 Successful download + +Test: + +- authenticated request; +- authorized access; +- storage object exists; +- streaming response; +- correct content type; +- correct content disposition; +- expected filename. + +## 9.3 Authorization + +Verify users cannot download files outside their permitted scope. + +## 9.4 Inline versus attachment + +Test CARE's policy for: + +```text +images +PDF +audio +video +documents +unknown types +``` + +The API SHALL set safe headers independent of provider-generated URLs. + +## 9.5 Filename safety + +Test filenames containing: + +- quotes; +- semicolons; +- newlines; +- non-ASCII characters; +- path separators; +- control characters. + +The response SHALL prevent header injection and path traversal. + +## 9.6 Streaming behavior + +Verify that the response does not read the complete file into memory. + +For provider integration tests, record memory behavior for representative file +sizes. + +## 9.7 Range requests + +If CARE or the frontend requires media seeking, test HTTP range behavior. + +Range support SHALL not be assumed automatically. + +If not supported in the initial implementation, the limitation SHALL be +documented. + +## 9.8 Missing object + +Verify a controlled response when the database record exists but the object is +missing. + +## 9.9 Storage timeout + +Simulate a slow or failed storage read. + +Verify: + +- request timeout behavior; +- controlled error response; +- logging without sensitive content. + +--- + +# 10. Task Dispatch Unit Tests + +## 10.1 Objective + +Verify backend-neutral task dispatch behavior. + +## 10.2 Common dispatch cases + +Test: + +```text +task name +JSON payload +delay +external task ID +configuration selection +serialization failure +backend failure +``` + +## 10.3 Payload restrictions + +Reject or clearly fail for: + +- model instances; +- querysets; +- file objects; +- unserializable objects; +- credentials; +- unexpected binary payloads. + +## 10.4 Transaction behavior + +Verify tasks scheduled through `transaction.on_commit` are: + +- not dispatched before commit; +- dispatched after commit; +- not dispatched after rollback. + +## 10.5 Unknown task + +The dispatcher or handler registry SHALL reject unknown task names. + +## 10.6 Backend selection + +Verify: + +```text +CARE_TASK_BACKEND=celery +CARE_TASK_BACKEND=cloud_tasks +``` + +and, if accepted: + +```text +CARE_TASK_BACKEND=postgres +``` + +select the expected implementation. + +--- + +# 11. Celery Compatibility Tests + +## 11.1 Objective + +Ensure local and traditional deployments continue working. + +## 11.2 Existing tasks + +Test existing wrappers for: + +```text +cleanup_expired_token_slots +cleanup_incomplete_file_uploads +generate_report_task +send_totp_enabled_email +send_totp_disabled_email +``` + +## 11.3 Reusable logic + +Verify Celery wrappers invoke the same reusable functions used by other +backends. + +## 11.4 Retry behavior + +Verify task-specific retry behavior remains compatible. + +## 11.5 Periodic registration + +Verify local Celery Beat registers the expected schedules. + +## 11.6 Result compatibility + +Where existing call sites use Celery task IDs or results, test that local +behavior remains unchanged until those call sites are deliberately refactored. + +--- + +# 12. Cloud Tasks Unit Tests + +## 12.1 Objective + +Verify Cloud Tasks request construction without requiring a real GCP queue. + +## 12.2 Client mocking + +Mock the official Cloud Tasks client. + +Verify: + +- parent queue path; +- HTTP method; +- worker URL; +- JSON body; +- content type; +- OIDC service account; +- OIDC audience; +- schedule time; +- generated task name. + +## 12.3 Delay + +Test: + +- no delay; +- positive delay; +- invalid negative delay; +- maximum supported delay according to project policy. + +## 12.4 Deterministic names + +If task IDs are used for deduplication, test: + +- valid name conversion; +- duplicate-name response; +- unsafe characters; +- length limits. + +## 12.5 Sensitive logging + +Verify task payloads are not fully logged. + +--- + +# 13. Cloud Tasks Worker Tests + +## 13.1 Objective + +Verify the private HTTP worker safely executes registered handlers. + +## 13.2 Method validation + +Only POST SHALL be accepted. + +## 13.3 Content-type validation + +Invalid content types SHALL be rejected. + +## 13.4 Authentication boundary + +Application tests MAY simulate authenticated and unauthenticated requests. + +Deployment smoke tests SHALL verify actual Cloud Run IAM behavior. + +## 13.5 Handler registry + +Test: + +- known task; +- unknown task; +- malformed task name; +- missing payload; +- invalid payload; +- extra fields. + +## 13.6 Arbitrary execution protection + +Verify a request cannot execute: + +- arbitrary Python paths; +- imported modules; +- `eval`; +- shell commands; +- unregistered callables. + +## 13.7 Success response + +A handler that completes successfully SHALL return a 2xx response. + +## 13.8 Transient failure + +A retriable exception SHALL produce a response that allows Cloud Tasks retry. + +## 13.9 Permanent failure + +A permanent validation or business error SHALL follow a defined non-retry or +limited-retry policy. + +## 13.10 Retry metadata + +Test parsing and logging of Cloud Tasks metadata headers when present. + +The headers SHALL not replace IAM authentication. + +--- + +# 14. Task Idempotency Tests + +## 14.1 Objective + +Ensure retries and duplicate delivery do not corrupt CARE state. + +## 14.2 Email tasks + +Test duplicate execution behavior. + +The project SHALL decide whether duplicate security-notification emails are: + +- acceptable; +- suppressed through idempotency; +- limited through task identity. + +## 14.3 Report generation + +Test repeated execution using the same logical request. + +Verify: + +- duplicate records are prevented or intentional; +- object names remain consistent with policy; +- partial previous attempts are handled; +- progress state is updated correctly. + +## 14.4 Cleanup tasks + +Execute cleanup commands repeatedly. + +The second execution SHALL succeed with no remaining matching records. + +## 14.5 Task execution records + +If an idempotency or execution table is introduced, test: + +- first claim; +- duplicate claim; +- completed task; +- failed task; +- retry; +- stale execution; +- concurrent requests. + +--- + +# 15. Cloud Run Job Tests + +## 15.1 Objective + +Verify administrative and scheduled commands can run as one-off jobs. + +## 15.2 Management commands + +Test: + +```text +migrate +sync_permissions_roles +sync_valueset +cleanup_expired_token_slots +cleanup_incomplete_file_uploads +``` + +## 15.3 Empty-state behavior + +On an empty database and empty buckets: + +- cleanup commands SHALL succeed; +- synchronization commands SHALL complete; +- commands SHALL not assume existing patients or files. + +## 15.4 Exit codes + +Successful jobs SHALL return exit code zero. + +Failures SHALL return non-zero. + +## 15.5 Repeated execution + +Management commands intended for scheduled use SHALL be safe to execute +repeatedly. + +## 15.6 Job timeout + +Long-running behavior SHALL be tested against configured Cloud Run Job limits. + +--- + +# 16. Cloud Scheduler Tests + +## 16.1 Objective + +Verify scheduled definitions correspond to intended CARE behavior. + +## 16.2 Configuration validation + +Terraform tests SHALL verify: + +- schedule expression; +- timezone; +- target job or queue; +- authentication; +- retry policy; +- enabled state. + +## 16.3 Schedule mapping + +Verify: + +```text +expired token cleanup -> daily +incomplete upload cleanup -> configured cadence +``` + +## 16.4 Duplicate scheduling + +The GCP profile SHALL not start Celery Beat. + +A configuration test SHALL prevent both: + +```text +Cloud Scheduler +Celery Beat +``` + +from being enabled for the same environment unless explicitly intended. + +--- + +# 17. PostgreSQL Cache Tests + +## 17.1 Objective + +Verify Django's database cache works as shared cache across instances. + +## 17.2 Basic cache operations + +Test: + +```text +set +get +add +delete +clear +timeout +expiration +get_many +set_many +``` + +Only operations actually required by CARE must be treated as mandatory. + +## 17.3 Cross-process visibility + +Use two separate Django processes or database connections. + +Verify a value written by one is visible to the other. + +## 17.4 Expiration + +Verify expired entries are treated as missing. + +The suite SHOULD also observe whether expired rows are removed according to +Django's database-cache behavior. + +## 17.5 Culling + +Configure a small maximum-entry count in a dedicated test. + +Verify culling does not produce application errors. + +## 17.6 Report progress + +Test progress updates through PostgreSQL cache: + +```text +initial value +progress update +read from another process +expiration +clear +``` + +## 17.7 Database failure + +Verify application behavior when the cache table is unavailable. + +The expected behavior SHALL be explicit. + +For example: + +```text +performance cache -> controlled miss or error policy +report progress -> controlled temporary failure +``` + +## 17.8 Query volume + +Performance tests SHALL measure database queries generated by cache use. + +The project SHALL identify endpoints where database-backed cache creates more +load than the uncached operation. + +--- + +# 18. PostgreSQL Rate-Limit Tests + +## 18.1 Objective + +Verify globally consistent limits across Cloud Run instances. + +## 18.2 Shared enforcement + +Simulate requests from separate application processes. + +Verify they contribute to the same limit. + +## 18.3 Concurrent requests + +Test simultaneous requests around the threshold. + +The implementation SHALL not allow excessive requests due to lost updates. + +## 18.4 Key dimensions + +Test limits based on the dimensions actually used by CARE, such as: + +```text +user +IP +endpoint +operation +facility +``` + +## 18.5 Window expiration + +Verify counters reset or expire as expected. + +## 18.6 Failure policy + +Test behavior during database errors. + +Security-sensitive limits SHALL have a documented fallback. + +--- + +# 19. PostgreSQL Transient-State Tests + +## 19.1 Objective + +Verify shared short-lived state stored in PostgreSQL behaves correctly. + +## 19.2 Explicit models + +For correctness-sensitive state, test: + +- creation; +- expiration; +- concurrent update; +- cleanup; +- audit fields; +- uniqueness. + +## 19.3 Cache-backed state + +For disposable state, reuse PostgreSQL cache tests. + +## 19.4 Cleanup + +Expired state SHALL be removable without affecting active records. + +--- + +# 20. Optional PostgreSQL Queue Tests + +This section applies only if the PostgreSQL task backend is accepted. + +## 20.1 Objective + +Verify reliable queue behavior without claiming serverless equivalence to Cloud +Tasks. + +## 20.2 Schema creation + +Test queue schema creation from an empty database. + +## 20.3 Enqueue and claim + +Test: + +- enqueue; +- single-worker claim; +- competing workers; +- no double claim; +- completion. + +## 20.4 Transactional enqueue + +Verify a queued task: + +- exists after transaction commit; +- does not exist after rollback. + +## 20.5 Retry + +Test: + +- transient failure; +- retry count; +- delay; +- terminal failure. + +## 20.6 Worker restart + +Verify jobs remain available after worker termination. + +## 20.7 Queue cleanup + +Test retention and removal of completed or failed jobs. + +## 20.8 Database pressure + +Measure: + +- connections; +- polling queries; +- notification behavior; +- queue-table growth; +- impact on CARE queries. + +## 20.9 Worker freshness + +Health checks SHALL detect an inactive or stale worker. + +--- + +# 21. Redis Compatibility Tests + +## 21.1 Objective + +Verify Redis can be enabled selectively and remains optional. + +## 21.2 Supported configurations + +Test: + +```text +Redis cache only +Redis rate limiting only +Redis transient state only +Celery + Redis +``` + +## 21.3 TLS + +For providers such as Upstash, test: + +```text +rediss:// +certificate verification +connection timeout +authentication +``` + +## 21.4 Command compatibility + +Test only commands actually required by CARE. + +Do not assume complete Redis command or module compatibility. + +## 21.5 Outage behavior + +Simulate Redis unavailability. + +Verify documented behavior for: + +```text +cache +rate limiting +progress +Celery broker +``` + +## 21.6 Redis-free startup + +The most important negative test is: + +```text +CARE starts and serves requests without any Redis URL +``` + +for the default GCP profile. + +--- + +# 22. Container Tests + +## 22.1 Objective + +Verify the production image works before cloud deployment. + +## 22.2 Image build + +The image SHALL build from a clean checkout. + +## 22.3 No secrets + +Inspect image layers and environment defaults for: + +- credentials; +- `.env` files; +- service-account JSON; +- database passwords; +- private keys. + +## 22.4 Runtime user + +Verify the process runs as the intended non-root user when configured. + +## 22.5 API startup + +Start the image locally with: + +```text +CARE_PROCESS_ROLE=api +``` + +Verify Gunicorn listens on `$PORT`. + +## 22.6 Worker startup + +Start the image in task-worker mode. + +Verify the internal task endpoint is reachable locally. + +## 22.7 Job command + +Run a management command using the same image. + +## 22.8 Static files + +Verify collected static assets are included and served. + +## 22.9 Redis-free image startup + +Start the GCP profile without Redis. + +No startup script SHALL wait for Redis. + +--- + +# 23. Terraform Tests + +## 23.1 Objective + +Verify infrastructure configuration before application deployment. + +## 23.2 Formatting and validation + +`terraform fmt` and `terraform validate` both act on a single directory. +`fmt` therefore needs `-recursive` to reach `modules/` and `environments/`, and +`validate` needs one invocation per initialized directory — there is no +recursive form. + +From `deploy/gcp/terraform/` (layout: 03-migration-plan.md §29): + +```bash +terraform fmt -check -recursive . + +for dir in environments/*/; do + terraform -chdir="$dir" init -backend=false + terraform -chdir="$dir" validate +done +``` + +Modules are validated through the environments that instantiate them. A module +that no environment references SHALL be validated on its own the same way. + +## 23.3 Plan review + +CI SHALL generate a plan for approved environments. + +The plan SHALL be reviewable before apply. + +## 23.4 Policy tests + +Verify: + +- buckets are private; +- worker disallows unauthenticated access; +- service accounts do not receive Owner or Editor; +- Cloud SQL is not unrestricted publicly; +- minimum instances follow environment policy; +- secrets are not stored as plain Terraform variables where avoidable; +- task invoker can invoke only the worker. + +## 23.5 Environment separation + +Verify development, staging and production names do not collide. + +## 23.6 Destructive changes + +CI SHOULD highlight: + +- Cloud SQL replacement; +- bucket deletion; +- secret deletion; +- service-account replacement; +- task-queue deletion. + +--- + +# 24. Cloud Run Deployment Smoke Tests + +## 24.1 Objective + +Verify the deployed development or staging environment works. + +## 24.2 API health + +Test the public or authenticated health endpoint. + +## 24.3 Database + +Execute a safe database-backed request. + +## 24.4 Static files + +Request a known static asset. + +## 24.5 File flow + +Using synthetic data: + +1. upload a small file through CARE; +2. confirm the record exists; +3. download it through CARE; +4. compare content; +5. delete it; +6. verify subsequent access fails. + +## 24.6 Task flow + +1. enqueue a test-safe task; +2. confirm worker invocation; +3. confirm expected database or email-test result; +4. verify task logs contain the task ID; +5. verify no sensitive payload is logged. + +## 24.7 Scale from zero + +After allowing services to scale down, invoke: + +```text +API +worker task +``` + +and verify cold-start behavior remains within acceptable limits. + +## 24.8 Unauthorized worker call + +Call the worker without valid IAM identity. + +The request SHALL fail. + +## 24.9 Scheduled job + +Invoke one scheduled job manually and verify successful completion. + +--- + +# 25. Empty-State Tests + +## 25.1 Objective + +Verify a completely new CARE installation works before data exists. + +## 25.2 Empty database + +Test: + +- initial API behavior; +- admin setup; +- first facility creation; +- first user creation; +- first patient creation; +- list endpoints returning empty collections. + +## 25.3 Empty buckets + +Test: + +- no failures when buckets contain no objects; +- cleanup jobs succeed; +- missing-file responses remain controlled. + +## 25.4 Initial value sets + +Verify setup commands populate required value sets from an empty database. + +## 25.5 Initial permissions + +Verify permission synchronization produces expected roles and permissions. + +## 25.6 First report + +Test report behavior before and after required templates are created. + +## 25.7 First scheduled execution + +Periodic cleanup SHALL succeed before any matching data exists. + +--- + +# 26. Security Tests + +## 26.1 Objective + +Verify the GCP adaptation does not weaken CARE security. + +## 26.2 File access + +Test horizontal and vertical authorization for files. + +## 26.3 Path traversal + +Attempt object names or filenames containing: + +```text +../ +absolute paths +encoded traversal +backslashes +``` + +Storage-relative names SHALL remain safe. + +## 26.4 Malicious filenames + +Test response-header injection and unsafe characters. + +## 26.5 Oversized requests + +Verify large requests are rejected before exhausting instance resources. + +## 26.6 Worker authentication + +Verify Cloud Run IAM protects the worker independently of application headers. + +## 26.7 Task-name injection + +Attempt arbitrary handler execution. + +## 26.8 Secret exposure + +Inspect: + +- logs; +- error responses; +- task payloads; +- environment output; +- health endpoints. + +Secrets SHALL not be exposed. + +## 26.9 Bucket access + +Verify buckets are not anonymously readable or writable. + +## 26.10 Service-account privilege + +Review effective IAM permissions. + +No runtime service account SHALL have project-wide Owner or Editor. + +## 26.11 Database exposure + +Verify Cloud SQL is not openly reachable from unrestricted networks. + +--- + +# 27. Performance Tests + +## 27.1 Objective + +Verify the low-cost architecture remains usable. + +## 27.2 API latency + +Measure representative CARE requests under: + +```text +warm instance +cold start +moderate concurrency +``` + +## 27.3 File upload + +Measure: + +- duration; +- memory; +- CPU; +- temporary-disk use; +- Cloud Run timeout margin. + +Use representative supported sizes. + +## 27.4 File download + +Measure: + +- first-byte latency; +- throughput; +- memory; +- concurrent streams. + +## 27.5 PostgreSQL cache + +Measure: + +- cache read latency; +- cache write latency; +- queries per request; +- impact on application transactions. + +## 27.6 Rate limiting + +Measure counter update overhead under concurrent requests. + +## 27.7 Cloud Tasks + +Measure: + +- enqueue latency; +- queue-to-start latency; +- cold worker start; +- execution time. + +## 27.8 Database connections + +Measure active connections during: + +- API concurrency; +- worker concurrency; +- jobs; +- database cache use. + +## 27.9 Acceptance thresholds + +Thresholds SHALL be documented after the initial benchmark. + +The project SHALL not invent arbitrary guarantees before measurement. + +--- + +# 28. Failure and Recovery Tests + +## 28.1 Cloud SQL interruption + +Verify: + +- controlled request failure; +- connection recovery; +- no corrupted state; +- health-check response. + +## 28.2 Cloud Storage denial + +Verify upload and download failures are controlled. + +## 28.3 Cloud Tasks enqueue failure + +Verify the API does not falsely report successful task creation. + +## 28.4 Worker failure + +Verify Cloud Tasks retries according to policy. + +## 28.5 Duplicate task + +Verify idempotency. + +## 28.6 Email failure + +Verify retry and terminal failure behavior. + +## 28.7 PostgreSQL cache table missing + +Verify startup or health diagnostics identify the problem clearly. + +## 28.8 Redis outage + +When Redis is optional, verify unrelated functionality continues when +appropriate. + +## 28.9 Cloud Run Job failure + +Verify non-zero exit status and alerting. + +--- + +# 29. Plugin Compatibility Tests + +## 29.1 Objective + +Verify required CARE plugins do not depend on removed behavior. + +## 29.2 Inventory + +For each required plugin, inspect: + +- Celery tasks; +- Redis use; +- `boto3`; +- signed URLs; +- custom file managers; +- health checks; +- startup hooks; +- migrations. + +## 29.3 Storage + +Required plugins SHALL either: + +- use Django Storage API; +- remain compatible with the selected storage profile; +- be explicitly unsupported. + +## 29.4 Tasks + +Plugin tasks SHALL be classified as: + +```text +Cloud Tasks compatible +Cloud Run Job compatible +Celery-only +unsupported +``` + +## 29.5 Test gate + +A plugin SHALL not be enabled in production until its required runtime behavior +has been tested. + +--- + +# 30. CI Test Stages + +The CI pipeline SHOULD use stages similar to: + +```text +Stage 1: formatting and lint +Stage 2: unit tests +Stage 3: Django application tests +Stage 4: PostgreSQL integration +Stage 5: MinIO storage integration +Stage 6: Redis compatibility +Stage 7: container build and tests +Stage 8: Terraform validation +Stage 9: optional deployed-environment smoke tests +``` + +GCS and Cloud Tasks live integration tests MAY run: + +- on protected branches; +- nightly; +- before release; +- against an isolated development project. + +They need not run for every untrusted pull request if credentials would be +exposed. + +--- + +# 31. Test Data Policy + +Tests SHALL use synthetic data. + +Production patient data SHALL not be used. + +Test fixtures SHALL avoid: + +- real names; +- real clinical records; +- real contact information; +- real identifiers. + +Uploaded test files SHALL contain synthetic or empty sample content. + +--- + +# 32. Test Isolation + +Each test run SHOULD use isolated: + +```text +database schema or database +storage prefix or bucket +task queue +cache namespace +``` + +Cloud integration tests SHALL clean up resources they create. + +Cleanup failure SHALL be reported. + +--- + +# 33. Test Configuration + +Test settings SHOULD provide backend overrides. + +Examples: + +```text +CARE_TASK_BACKEND=fake +CARE_CACHE_BACKEND=dummy +``` + +for fast unit tests. + +`fake` is a **test-only** value. The production backend contract accepts +`cloud_tasks`, `celery` and `postgres` only, and production settings SHALL +reject `fake` at startup rather than silently discarding every task. It is +listed here because test settings are the one place it is permitted. + +Integration suites SHALL deliberately select: + +```text +celery +cloud_tasks +postgres +redis +``` + +as applicable. + +A fake backend SHALL not replace real integration testing. + +--- + +# 34. Coverage Expectations + +Coverage SHALL prioritize: + +- storage flows; +- authentication and authorization; +- task dispatch; +- task handlers; +- idempotency; +- cache selection; +- Redis optionality; +- settings validation; +- management commands; +- health checks. + +A single global coverage percentage SHALL not be treated as proof of adequate +testing. + +Critical migration paths SHALL have explicit tests. + +--- + +# 35. Release Gates + +A GCP release SHALL require: + +- upstream CARE tests pass; +- GCP settings tests pass; +- production image builds; +- Terraform validates; +- storage tests pass; +- file API security tests pass; +- task dispatcher tests pass; +- Cloud Tasks worker tests pass; +- PostgreSQL cache tests pass; +- Redis-free startup passes; +- deployed smoke tests pass in staging; +- no critical security findings remain. + +If the PostgreSQL queue is supported, its queue tests SHALL also pass for +profiles that use it. + +--- + +# 36. Upstream Synchronization Gates + +After merging a new upstream version, run: + +```text +local regression suite +storage integration suite +task suite +PostgreSQL cache suite +container build +Terraform validation +GCP smoke tests +``` + +Special review SHALL focus on upstream changes to: + +```text +settings +tasks +file models +file endpoints +Dockerfiles +startup scripts +health checks +plugins +``` + +--- + +# 37. Test Deliverables + +The implementation SHALL produce: + +```text +documented test commands +profile-specific test configuration +storage integration tests +file API tests +task dispatch tests +Cloud Tasks worker tests +task idempotency tests +PostgreSQL cache tests +rate-limit tests +optional Redis tests +container tests +Terraform validation +Cloud Run smoke tests +empty-state tests +security tests +upstream regression gates +``` + +--- + +# 38. Definition of Testing Completion + +Testing for the initial GCP implementation is complete when: + +- a clean local checkout passes the supported local workflow; +- CARE starts with GCP settings and no Redis; +- an empty Cloud SQL database initializes successfully; +- MinIO and GCS pass required storage tests; +- all production file traffic passes through tested Django endpoints; +- unauthorized file access is rejected; +- Cloud Tasks dispatch and private worker execution pass; +- duplicate task delivery is safe; +- Cloud Scheduler and Jobs execute successfully; +- PostgreSQL cache works across instances; +- rate limiting is shared and concurrency-safe; +- optional Redis works when enabled; +- the production image passes container tests; +- Terraform passes validation and policy checks; +- staging smoke tests pass from an empty environment; +- upstream synchronization runs the complete regression gate. + +--- + +## 39. Next Document + +The next document is: + +```text +docs/xii/architecture/05-upstream-sync.md +``` + +It will define: + +- remote and branch configuration; +- upstream update workflow; +- conflict-resolution rules; +- recurring-conflict analysis; +- test requirements after synchronization; +- rules for keeping `develop` clean; +- rules for maintaining the `gcp` branch; +- release tagging and rollback references. diff --git a/docs/xii/architecture/05-upstream-sync.md b/docs/xii/architecture/05-upstream-sync.md new file mode 100644 index 0000000000..921056fd94 --- /dev/null +++ b/docs/xii/architecture/05-upstream-sync.md @@ -0,0 +1,1206 @@ +--- +title: Upstream Synchronization +document: 05-upstream-sync +version: 0.1.0 +status: Draft +source_repository: https://github.com/ohcnetwork/care +upstream_repository: https://github.com/ohcnetwork/care +upstream_branch: develop +integration_branch: gcp +depends_on: + - docs/xii/architecture/00-scope-and-goals.md + - docs/xii/architecture/01-current-runtime.md + - docs/xii/architecture/02-target-runtime.md + - docs/xii/architecture/03-migration-plan.md + - docs/xii/architecture/04-testing.md +--- + +# Upstream Synchronization + +## 1. Purpose + +This document defines how the maintained GCP fork of CARE incorporates changes +from the official upstream repository. + +The synchronization process SHALL preserve: + +- a clean upstream mirror; +- a deployable GCP integration branch; +- traceable conflict resolution; +- reproducible tests; +- minimal divergence; +- the ability to identify which changes belong to upstream and which belong to + the GCP adaptation. + +This document assumes the official upstream development branch is: + +```text +ohcnetwork/care:develop +``` + +The fork's maintained GCP branch is: + +```text +origin/gcp +``` + +--- + +## 2. Core Rule + +The fork SHALL maintain a branch that mirrors upstream without local product +changes. + +The recommended branch is: + +```text +origin/develop +``` + +This branch SHALL contain the same history and content as: + +```text +upstream/develop +``` + +after each synchronization. + +GCP-specific commits SHALL NOT be added directly to `origin/develop`. + +--- + +## 3. Repository Remotes + +A working clone SHALL define at least: + +```text +origin +upstream +``` + +### 3.1 `origin` + +`origin` points to the maintained fork. + +Example: + +```bash +git remote add origin git@github.com:YOUR-ORGANIZATION/care.git +``` + +### 3.2 `upstream` + +`upstream` points to the official CARE repository. + +```bash +git remote add upstream https://github.com/ohcnetwork/care.git +``` + +### 3.3 Verification + +Run: + +```bash +git remote -v +``` + +Expected conceptual output: + +```text +origin git@github.com:YOUR-ORGANIZATION/care.git +upstream https://github.com/ohcnetwork/care.git +``` + +The exact transport MAY be SSH or HTTPS. + +--- + +## 4. Branch Model + +The recommended permanent branches are: + +```text +develop +gcp +``` + +Temporary branches include: + +```text +feature/* +fix/* +sync/upstream-YYYY-MM-DD +release/* +``` + +The conceptual flow is: + +```mermaid +flowchart TD + UPSTREAM[upstream/develop] --> MIRROR[origin/develop] + MIRROR --> SYNC[sync/upstream-YYYY-MM-DD] + GCP[origin/gcp] --> SYNC + SYNC --> GCP + GCP --> FEATURE[feature branches] + FEATURE --> GCP + GCP --> RELEASE[release tags or branches] +``` + +--- + +## 5. Branch Responsibilities + +### 5.1 `develop` + +Purpose: + +```text +exact or near-exact mirror of upstream/develop +``` + +Rules: + +- SHALL contain no GCP-specific commits; +- SHALL not contain local deployment patches; +- SHALL not contain environment secrets; +- SHALL not be used for production deployment; +- MAY be force-updated to match upstream; +- SHOULD be protected from normal pull-request merges. + +### 5.2 `gcp` + +Purpose: + +```text +maintained integration of CARE and the GCP adaptation +``` + +Rules: + +- SHALL contain all accepted GCP changes; +- SHOULD remain deployable; +- SHALL pass the supported test matrix; +- SHALL receive upstream changes through synchronization branches; +- SHOULD not be force-pushed after it becomes shared; +- SHOULD use pull requests for nontrivial changes. + +### 5.3 `feature/*` + +Purpose: + +```text +isolated implementation work +``` + +Rules: + +- SHOULD branch from `gcp`; +- SHOULD contain one coherent change; +- SHOULD be merged through pull request; +- SHALL run relevant tests; +- SHOULD be deleted after merge. + +### 5.4 `sync/upstream-YYYY-MM-DD` + +Purpose: + +```text +merge and validate a specific upstream update +``` + +Rules: + +- SHALL branch from the current `gcp`; +- SHALL merge the updated `develop`; +- SHALL contain conflict resolutions; +- SHALL contain only changes necessary for synchronization; +- SHALL pass local and GCP tests; +- SHALL be merged into `gcp` through pull request. + +### 5.5 `release/*` + +Optional purpose: + +```text +stabilization before a production release +``` + +Release branches MAY be omitted if immutable tags on `gcp` are sufficient. + +--- + +## 6. Initial Setup + +For a new fork, configure the upstream mirror. + +```bash +git remote add upstream https://github.com/ohcnetwork/care.git +git fetch upstream +``` + +Create or reset the local `develop` branch: + +```bash +git switch -C develop upstream/develop +``` + +Push the mirror: + +```bash +git push --force-with-lease origin develop +``` + +Create the initial GCP branch: + +```bash +git switch -c gcp +git push -u origin gcp +``` + +After GCP-specific development begins, `gcp` SHALL not be reset to upstream. + +--- + +## 7. Routine Upstream Synchronization + +The recommended synchronization workflow is: + +```bash +git fetch upstream +git fetch origin +``` + +Update the upstream mirror: + +```bash +git switch develop +git reset --hard upstream/develop +git push --force-with-lease origin develop +``` + +Create a synchronization branch from the current GCP integration: + +```bash +git switch gcp +git pull --ff-only origin gcp +git switch -c sync/upstream-YYYY-MM-DD +``` + +Merge the updated mirror: + +```bash +git merge develop +``` + +Resolve conflicts, run tests and open a pull request into: + +```text +gcp +``` + +--- + +## 8. Why Merge Instead of Rebase + +The maintained `gcp` branch SHOULD normally incorporate upstream through merge. + +Example: + +```bash +git merge develop +``` + +Reasons: + +- preserves the history of upstream synchronization events; +- avoids rewriting shared GCP history; +- makes conflict-resolution commits traceable; +- makes deployed revisions easier to audit; +- avoids forcing collaborators to repair rebased branches. + +Rebase MAY be used on private feature branches before merge. + +Rebase SHOULD NOT normally rewrite the shared `gcp` branch. + +--- + +## 9. Mirror Update Safety + +Updating `develop` uses: + +```bash +git reset --hard upstream/develop +``` + +and: + +```bash +git push --force-with-lease origin develop +``` + +This is acceptable only because `develop` is designated as an upstream mirror. + +`git reset --hard` discards the **local** `develop` and its uncommitted changes. +The pre-flight checks SHALL therefore inspect the local branch, not `origin`: + +```bash +git fetch upstream + +# 1. Nothing uncommitted is about to be destroyed. +git status --porcelain # SHALL be empty + +# 2. No local-only commits are about to be destroyed. +git log --oneline upstream/develop..develop # SHALL be empty +``` + +Checking `upstream/develop..origin/develop` is not sufficient: it describes what +the remote carries, while the reset acts on the local branch. A commit made +locally and never pushed is invisible to that comparison and would be lost. + +If local commits exist, they SHALL be: + +- moved to an appropriate feature branch; +- reviewed; +- removed from `develop`. + +The process SHALL not silently discard valuable work. + +After pushing, confirm the mirror is actually a mirror. `git log` in one +direction only proves the absence of extra commits, not equality of content: + +```bash +git diff --exit-code upstream/develop develop +git diff --exit-code upstream/develop origin/develop +``` + +Both SHALL exit zero. + +--- + +## 10. Synchronization Frequency + +Upstream synchronization SHOULD occur: + +- before beginning a major GCP feature; +- before a production release; +- after significant upstream security fixes; +- after upstream changes to storage, tasks, settings or deployment; +- regularly enough to prevent a very large divergence. + +A practical cadence MAY be: + +```text +monthly +before each release +immediately for relevant security updates +``` + +The exact cadence depends on upstream activity and deployment needs. + +Frequent small synchronizations are preferred over rare large ones. + +--- + +## 11. Pre-Merge Review + +Before merging `develop` into a synchronization branch, inspect upstream +changes. + +Useful commands: + +```bash +git log --oneline gcp..develop +``` + +```bash +git diff --stat gcp...develop +``` + +```bash +git diff --name-status gcp...develop +``` + +Review changes affecting: + +```text +config/settings/ +config/celery_app.py +care/emr/tasks/ +care/emr/utils/file_manager.py +file models and serializers +upload and download endpoints +Dockerfiles +Compose files +startup scripts +health checks +dependencies +plugins +migrations +authentication +permissions +``` + +The synchronization pull request SHOULD summarize relevant upstream changes. + +--- + +## 12. Conflict Resolution Principles + +Conflicts SHALL be resolved according to these priorities: + +1. preserve upstream business behavior; +2. preserve security fixes; +3. preserve data-model and migration correctness; +4. preserve the GCP deployment contract; +5. minimize custom code; +6. move deployment-specific behavior into isolated files where possible. + +The resolver SHALL not blindly select: + +```text +ours +``` + +or: + +```text +theirs +``` + +for entire files without understanding both sides. + +--- + +## 13. Conflict Categories + +Every conflict SHOULD be classified. + +Recommended categories: + +```text +settings +dependencies +storage +tasks +cache or Redis +Docker or runtime scripts +health checks +Terraform or CI +tests +documentation +unrelated application behavior +``` + +The synchronization pull request SHOULD identify: + +- affected files; +- conflict category; +- resolution; +- behavior preserved; +- tests run. + +--- + +## 14. Settings Conflicts + +Settings files are likely conflict areas. + +Preferred strategy: + +- preserve upstream `base.py` changes; +- preserve upstream `deployment.py` changes; +- keep GCP-specific overrides in `config/settings/gcp.py`; +- avoid copying large blocks from upstream settings into `gcp.py`; +- import and override only what differs. + +When upstream adds a new setting: + +1. determine whether the GCP profile can inherit it; +2. override only if GCP behavior differs; +3. add a GCP-specific test when required. + +Repeated settings conflicts indicate that too much GCP logic is located in +upstream-owned settings files. + +--- + +## 15. Dependency Conflicts + +When upstream changes dependency versions or lockfiles: + +1. preserve upstream dependency changes; +2. reapply GCP-specific dependencies using the repository's package manager; +3. regenerate the lockfile; +4. run the complete dependency test suite; +5. verify supported Python and Django versions; +6. verify `django-storages`, Google clients and task clients remain compatible. + +Do not manually combine lockfile sections without using the package manager. + +GCP dependencies SHOULD remain as small as practical. + +--- + +## 16. Storage Conflicts + +Storage-related upstream changes require special review. + +Inspect whether upstream has changed: + +- file models; +- object naming; +- upload completion semantics; +- MIME validation; +- cleanup behavior; +- file permissions; +- signed URL APIs; +- `files_manager`; +- bucket configuration. + +The GCP fork SHALL preserve the target storage policy: + +```text +Django Storage API +django-storages +server-mediated uploads +server-mediated downloads +no direct browser bucket access +``` + +If upstream adopts Django Storage API, the fork SHOULD remove redundant custom +patches and move closer to upstream. + +If upstream changes file behavior, the fork SHALL update storage and API tests +before merging. + +--- + +## 17. Task Conflicts + +Inspect upstream changes to: + +- Celery task signatures; +- task names; +- retries; +- schedules; +- call sites; +- result usage; +- task modules; +- plugin tasks. + +The GCP fork SHALL preserve: + +```text +Cloud Tasks as default GCP backend +Celery compatibility locally +reusable task logic +explicit handler registration +``` + +When upstream adds a new Celery task: + +1. keep it working under Celery; +2. classify it; +3. decide whether GCP uses Cloud Tasks, Cloud Run Jobs or synchronous execution; +4. add it to the task inventory; +5. add tests before enabling it in production. + +Upstream task names SHOULD remain stable where possible. + +--- + +## 18. Cache and Redis Conflicts + +When upstream adds new default-cache or Redis use: + +1. identify the responsibility; +2. determine whether it requires shared state; +3. determine whether PostgreSQL is sufficient; +4. determine whether Redis remains optional; +5. add backend-specific tests; +6. avoid making `REDIS_URL` mandatory in GCP unintentionally. + +New upstream Redis usage SHALL be classified as: + +```text +cache +rate limiting +progress +lock +temporary state +session +Celery +direct Redis use +unknown +``` + +Unknown use SHALL not be merged into production without investigation. + +--- + +## 19. Docker and Startup Conflicts + +Upstream may change: + +- development Dockerfile; +- production Dockerfile; +- Compose services; +- startup scripts; +- migration behavior; +- Celery startup; +- health checks. + +The GCP fork SHOULD preserve upstream local development behavior. + +GCP runtime behavior SHOULD remain isolated in: + +```text +docker/gcp.Dockerfile +scripts/start-gcp-api.sh +scripts/start-gcp-task-worker.sh +scripts/run-gcp-job.sh +config/settings/gcp.py +deploy/gcp/ +``` + +If upstream provides a production container suitable for Cloud Run, the fork +SHOULD evaluate reusing it rather than maintaining a duplicate image. + +--- + +## 20. Database Migration Conflicts + +Upstream migrations SHALL generally be accepted unchanged. + +The GCP fork SHALL NOT edit upstream migration files after release merely to +resolve conflicts. + +After synchronization, test from an empty database: + +```bash +python manage.py migrate --noinput +``` + +Because production is greenfield initially, clean-schema migration is a primary +test. + +After real production use begins, also test upgrades from the currently +deployed schema. + +Migration conflicts involving plugins SHALL be resolved according to plugin +ownership and documented dependencies. + +--- + +## 21. Test Conflicts + +Upstream tests SHALL be preserved. + +When upstream changes expected behavior, GCP-specific tests SHALL be reviewed +for assumptions that are no longer valid. + +The fork SHALL not weaken upstream assertions merely to make GCP patches pass. + +If a GCP-specific test conflicts with an upstream behavior change, determine +whether: + +- the upstream change should be inherited; +- the GCP implementation should adapt; +- the GCP deployment policy intentionally differs; +- an ADR is required. + +--- + +## 22. Documentation Conflicts + +GCP documentation MAY refer to paths, task names, settings or commands that +upstream changes. + +After synchronization, search documentation for outdated references. + +Examples: + +```bash +grep -R "S3FilesManager" docs/xii +grep -R "CELERY_BROKER_URL" docs/xii +grep -R "config.settings.deployment" docs/xii +``` + +Documentation changes SHALL be included in the synchronization pull request +when implementation changes affect them. + +--- + +## 23. Required Test Sequence + +After conflict resolution, run the local upstream-compatible tests first. + +```bash +make build +make up +make load-fixtures +make test +``` + +Then run GCP-specific tests. + +Required conceptual groups: + +```text +GCP settings +production image build +Cloud SQL migration from empty schema +Django storage aliases +MinIO integration +GCS integration +file upload API +file download API +task dispatch +Cloud Tasks worker +PostgreSQL cache +Redis-free startup +optional Redis profile +Cloud Run Jobs +Terraform validation +``` + +The exact commands SHALL be documented in `04-testing.md` and project scripts. + +--- + +## 24. Synchronization Pull Request Template + +A synchronization pull request SHOULD include: + +```markdown +## Upstream range + +Previous upstream commit: +`` + +New upstream commit: +`` + +## Relevant upstream changes + +- ... +- ... + +## Conflicts resolved + +| File | Category | Resolution | +|---|---|---| +| ... | ... | ... | + +## GCP adaptations updated + +- ... +- ... + +## Tests + +- [ ] Local build +- [ ] Local test suite +- [ ] GCP settings +- [ ] Storage integration +- [ ] Task tests +- [ ] PostgreSQL cache +- [ ] Container build +- [ ] Terraform validation +- [ ] Staging smoke tests + +## Known follow-up + +- ... +``` + +--- + +## 25. Synchronization Commit Style + +A synchronization branch MAY contain: + +1. the upstream merge commit; +2. focused conflict-resolution commits; +3. test or documentation fixes required by upstream changes. + +Recommended commit examples: + +```text +merge: sync upstream develop 2026-08-05 +fix(storage): adapt file API to upstream upload model changes +fix(tasks): register new upstream notification task +test(gcp): update storage regression cases +docs(gcp): update settings references after upstream sync +``` + +Avoid mixing unrelated new GCP features into the synchronization branch. + +--- + +## 26. Release References + +Each production release SHOULD record: + +- GCP integration commit SHA; +- upstream base commit SHA; +- container image digest; +- Terraform commit SHA; +- database migration state; +- deployment timestamp. + +A release tag MAY use: + +```text +gcp-vYYYY.MM.DD.N +``` + +or semantic versioning. + +The tag message SHOULD include: + +```text +Upstream base: +GCP commit: +Image digest: +``` + +--- + +## 27. Tracking the Upstream Base + +The repository SHOULD make the upstream base easy to identify. + +Possible mechanisms: + +- merge history; +- release notes; +- a text file such as `UPSTREAM_BASE`; +- build metadata; +- deployment annotations. + +Example file: + +```text +UPSTREAM_BASE +``` + +Contents: + +```text +repository=https://github.com/ohcnetwork/care +branch=develop +commit= +synchronized_at= +``` + +If this file is used, it SHALL be updated only during upstream synchronization. + +--- + +## 28. Recurring Conflict Register + +The project SHOULD maintain: + +```text +docs/xii/architecture/upstream-conflicts.md +``` + +For each recurring conflict, record: + +```text +file +reason +frequency +current workaround +preferred structural improvement +upstream PR possibility +``` + +Example: + +```markdown +## config/settings/base.py + +Reason: +GCP cache selection currently modifies the shared cache block. + +Preferred improvement: +Move all GCP cache overrides into `config/settings/gcp.py`. + +Status: +Open. +``` + +The register helps reduce future maintenance cost. + +--- + +## 29. Reducing Divergence + +When a conflict recurs, prefer these remedies in order: + +1. move GCP behavior into `config/settings/gcp.py`; +2. add a new GCP-specific script; +3. add a helper module; +4. use an existing upstream extension point; +5. propose a small provider-neutral upstream improvement; +6. maintain a focused patch only when necessary. + +The project SHALL not respond to recurring conflicts by copying entire upstream +modules into GCP-specific versions unless there is no reasonable alternative. + +--- + +## 30. Upstream Contributions + +Some GCP work may be suitable for upstream contribution. + +Potential upstream-friendly improvements include: + +- using Django Storage API; +- reducing direct `boto3` coupling; +- extracting reusable task logic; +- making Redis health checks conditional; +- avoiding unconditional Redis startup dependencies; +- using management commands for periodic cleanup; +- improving backend configuration validation. + +An upstream pull request SHOULD: + +- remain provider-neutral; +- preserve existing behavior; +- avoid GCP-specific names; +- include tests; +- be small enough to review independently. + +GCP-specific Terraform, IAM and Cloud Run definitions generally remain in the +fork unless upstream requests them. + +--- + +## 31. Security Update Process + +Relevant upstream security changes SHALL be prioritized. + +When upstream publishes or merges a security fix: + +1. inspect the affected code; +2. determine whether the GCP fork modifies the same area; +3. create an expedited synchronization branch; +4. resolve conflicts carefully; +5. run focused security tests; +6. deploy an immutable release; +7. record the upstream base. + +Security fixes SHALL not wait for the normal synchronization cadence when they +affect deployed functionality. + +--- + +## 32. Failed Synchronization + +If a synchronization cannot be completed safely: + +- do not merge partially resolved code into `gcp`; +- document the blocking upstream change; +- keep the current production revision; +- create focused investigation branches; +- identify whether a GCP customization must be redesigned; +- avoid making unsupported claims of compatibility. + +The synchronization pull request MAY remain draft until tests pass. + +--- + +## 33. Reverting a Synchronization + +If a merged synchronization causes application regressions, prefer reverting +the synchronization merge or deploying the previous immutable release. + +Example: + +```bash +git revert -m 1 +``` + +The correct parent number SHALL be verified before executing the revert. + +Do not reset shared `gcp` history after deployment. + +A Git revert does not automatically reverse database migrations. + +Database compatibility SHALL be evaluated separately. + +--- + +## 34. First Greenfield Release + +Before the first real production use, upstream synchronization is simpler +because there is no production data to preserve. + +The project MAY: + +- recreate development and staging environments; +- rerun all migrations from an empty schema; +- rebuild empty buckets; +- replace experimental infrastructure. + +The first production release SHALL still record: + +- exact upstream commit; +- exact GCP commit; +- image digest; +- Terraform state; +- test results. + +--- + +## 35. Synchronization After Real Use Begins + +After real patient or operational data exists, synchronization SHALL include: + +- database upgrade testing; +- backwards-compatible migration review; +- production backup verification; +- storage API regression testing; +- application rollback compatibility; +- scheduled-job review; +- task idempotency review. + +The project SHALL no longer treat production resources as disposable. + +--- + +## 36. Automated Upstream Monitoring + +The project MAY automate detection of new upstream commits. + +Possible mechanisms: + +- scheduled GitHub Action; +- repository comparison workflow; +- dependency update bot; +- release-monitoring task. + +Automation MAY open an issue or draft pull request. + +It SHALL not automatically merge upstream changes into `gcp` without tests and +review. + +--- + +## 37. Suggested Automation Workflow + +A scheduled workflow MAY: + +1. fetch upstream; +2. compare `origin/develop` with `upstream/develop`; +3. report new commits; +4. list changed files; +5. flag high-risk paths; +6. open an issue or draft synchronization pull request. + +High-risk paths include: + +```text +config/settings/ +care/emr/tasks/ +care/emr/utils/ +file models +upload endpoints +download endpoints +Dockerfiles +scripts/ +migrations/ +requirements or lockfiles +``` + +--- + +## 38. Synchronization Checklist + +Before opening a synchronization pull request: + +- [ ] Fetch `origin` and `upstream`. +- [ ] Verify local worktree is clean. +- [ ] Reset `develop` to `upstream/develop`. +- [ ] Push `origin/develop` with `--force-with-lease`. +- [ ] Create a dated synchronization branch from `gcp`. +- [ ] Review upstream commit range. +- [ ] Merge `develop`. +- [ ] Classify every conflict. +- [ ] Resolve conflicts without discarding security or domain changes. +- [ ] Update GCP settings and documentation where required. +- [ ] Run local tests. +- [ ] Run GCP tests. +- [ ] Build the production image. +- [ ] Validate Terraform. +- [ ] Run staging smoke tests for high-risk updates. +- [ ] Record the new upstream base. + +--- + +## 39. Merge Checklist + +Before merging into `gcp`: + +- [ ] Pull request review complete. +- [ ] No unresolved conflict markers. +- [ ] No accidental secrets. +- [ ] Local profile passes. +- [ ] Redis-free GCP profile passes. +- [ ] Storage tests pass. +- [ ] Task tests pass. +- [ ] PostgreSQL cache tests pass. +- [ ] Optional Redis tests pass when relevant. +- [ ] Empty-database migrations pass. +- [ ] Documentation references are current. +- [ ] Upstream base metadata is updated. +- [ ] Deployment risk is understood. + +--- + +## 40. Definition of Successful Synchronization + +An upstream synchronization is complete when: + +- `origin/develop` matches `upstream/develop`; +- the new upstream code is merged into `gcp`; +- conflicts are documented; +- local CARE behavior passes tests; +- the GCP runtime passes tests; +- a production image builds; +- Terraform validates; +- the upstream base is recorded; +- no known security or data-model regression remains; +- the `gcp` branch is deployable. + +--- + +## 41. Next Document + +The next document is: + +```text +docs/xii/architecture/06-operations.md +``` + +It will define: + +- initial environment creation; +- deployment; +- database initialization; +- secret management; +- Cloud Run service operation; +- Cloud Run Jobs; +- Cloud Scheduler; +- task and queue operations; +- storage operations; +- PostgreSQL cache maintenance; +- optional Redis operation; +- backups; +- recovery; +- monitoring; +- cost controls; +- routine maintenance. diff --git a/docs/xii/architecture/06-operations.md b/docs/xii/architecture/06-operations.md new file mode 100644 index 0000000000..652631c3cb --- /dev/null +++ b/docs/xii/architecture/06-operations.md @@ -0,0 +1,2549 @@ +--- +title: GCP Operations Guide +document: 06-operations +version: 0.1.0 +status: Draft +source_repository: https://github.com/ohcnetwork/care +target_platform: Google Cloud Platform +deployment_type: Greenfield +depends_on: + - docs/xii/architecture/00-scope-and-goals.md + - docs/xii/architecture/01-current-runtime.md + - docs/xii/architecture/02-target-runtime.md + - docs/xii/architecture/03-migration-plan.md + - docs/xii/architecture/04-testing.md + - docs/xii/architecture/05-upstream-sync.md +--- + +# GCP Operations Guide + +## 1. Purpose + +This document defines how to create, deploy, operate, observe, maintain and +recover a greenfield CARE environment on Google Cloud Platform. + +It covers the operational lifecycle after the architecture and implementation +described in the preceding documents have been completed. + +The guide assumes: + +- CARE runs on Cloud Run; +- PostgreSQL runs on Cloud SQL; +- files use Django Storage API and Cloud Storage; +- asynchronous work uses Cloud Tasks by default; +- task execution occurs in a private Cloud Run service; +- periodic work uses Cloud Scheduler and Cloud Run Jobs; +- PostgreSQL may provide cache, rate limiting and transient state; +- Redis-compatible services remain optional; +- infrastructure is managed using Terraform; +- the same application image supports API, worker and job roles. + +This guide does not describe the original Docker Compose runtime except where +it is used for local development or troubleshooting. + +--- + +# 2. Operational Objectives + +Operations SHALL prioritize: + +1. patient-data protection; +2. correctness; +3. recoverability; +4. service availability; +5. observability; +6. upstream maintainability; +7. predictable cost; +8. operational simplicity. + +Cost reductions SHALL not disable: + +- database backups; +- access controls; +- required logs; +- recovery procedures; +- application health checks. + +--- + +# 3. Environments + +The supported environments SHOULD be: + +```text +dev +staging +prod +``` + +Each environment SHOULD have separate: + +- Cloud Run services; +- Cloud Run Jobs; +- Cloud SQL database or instance; +- Cloud Storage buckets; +- Cloud Tasks queues; +- Cloud Scheduler jobs; +- secrets; +- service-account identities; +- monitoring labels; +- deployment history. + +Production data SHALL NOT be copied into development or staging without an +approved anonymization process. + +--- + +# 4. Naming Convention + +Resources SHOULD follow a predictable naming convention. + +Example: + +```text +care-- +``` + +Examples: + +```text +care-dev-api +care-dev-worker +care-dev-migrate +care-dev-cleanup-files +care-dev-tasks +care-dev-patient-files +care-dev-facility-files +care-dev-reports +care-dev-db +``` + +Production examples: + +```text +care-prod-api +care-prod-worker +care-prod-migrate +care-prod-tasks +``` + +Names SHOULD identify: + +- application; +- environment; +- role. + +Names SHOULD NOT contain: + +- patient information; +- organization secrets; +- credentials; +- temporary developer names in production. + +--- + +# 5. Resource Labels + +All supported GCP resources SHOULD use labels or annotations containing: + +```text +application=care +environment=dev|staging|prod +managed_by=terraform +component=api|worker|database|storage|tasks|jobs +``` + +Deployed Cloud Run revisions SHOULD also expose: + +```text +APP_VERSION +GIT_COMMIT_SHA +UPSTREAM_COMMIT_SHA +DEPLOYED_AT +``` + +These values improve: + +- log filtering; +- cost analysis; +- incident investigation; +- release traceability. + +--- + +# 6. Initial Project Preparation + +Before creating CARE infrastructure: + +1. Select or create the GCP project. +2. Configure billing. +3. establish administrative ownership; +4. configure Terraform state storage; +5. configure deployment identity; +6. determine the primary region; +7. determine data-location requirements; +8. define environment naming; +9. define backup and retention requirements; +10. define production access procedures. + +The primary region SHOULD be selected deliberately. + +The API, worker, Cloud SQL, Cloud Storage and task queue SHOULD normally be +placed in compatible nearby locations to reduce: + +- latency; +- network cost; +- operational complexity. + +--- + +# 7. Required APIs + +Terraform SHOULD enable only the APIs required by the deployment. + +The expected set includes APIs for: + +```text +Cloud Run +Cloud SQL +Cloud Storage +Cloud Tasks +Cloud Scheduler +Artifact Registry +Secret Manager +Cloud Logging +Cloud Monitoring +IAM +service networking where required +``` + +The exact API identifiers SHALL live in Terraform rather than this operational +document. + +API enablement SHALL be reproducible. + +Operators SHOULD NOT rely on manually enabled APIs that are absent from +infrastructure code. + +--- + +# 8. Terraform State + +Terraform state SHALL be stored remotely. + +The state backend SHALL: + +- restrict access; +- enable versioning where supported; +- prevent anonymous access; +- separate environments; +- support recovery of previous state versions; +- avoid storage in developer laptops as the authoritative copy. + +Recommended conceptual separation: + +```text +terraform-state/ +├── dev/ +├── staging/ +└── prod/ +``` + +Production Terraform access SHALL be more restrictive than development access. + +The Terraform state may contain sensitive infrastructure metadata. + +It SHALL be treated as confidential. + +--- + +# 9. Terraform Workflow + +The normal workflow is: + +```bash +terraform fmt -check +terraform init +terraform validate +terraform plan +terraform apply +``` + +Production changes SHOULD use a reviewed plan. + +The reviewed plan SHALL correspond to the exact code later applied. + +Operators SHALL inspect plans for: + +- Cloud SQL replacement; +- bucket deletion; +- service-account replacement; +- secret deletion; +- Cloud Run service recreation; +- queue deletion; +- IAM broadening; +- network changes. + +Destructive stateful changes SHALL not be applied casually. + +--- + +# 10. Environment Creation Order + +A new environment SHOULD be created in this order: + +1. project APIs; +2. Artifact Registry; +3. service accounts; +4. Secret Manager resources; +5. networking; +6. Cloud SQL; +7. Cloud Storage buckets; +8. Cloud Tasks queues; +9. container image; +10. Cloud Run Jobs; +11. database initialization; +12. worker service; +13. API service; +14. Cloud Scheduler; +15. smoke tests; +16. alerts and dashboards. + +The exact Terraform dependency graph MAY automate much of this order. + +Application initialization SHALL still remain explicit. + +--- + +# 11. Service Accounts + +The recommended service accounts are: + +```text +care-api +care-worker +care-jobs +care-tasks-invoker +care-deployer +``` + +A smaller development environment MAY consolidate identities temporarily. + +Production SHOULD keep responsibilities separate. + +--- + +# 12. API Service Account + +The API identity SHOULD receive only permissions required to: + +- connect to Cloud SQL; +- read required secrets; +- read and write configured CARE buckets; +- enqueue Cloud Tasks; +- emit logs and metrics; +- call explicitly required external services. + +The API identity SHALL NOT receive: + +- project Owner; +- project Editor; +- unrestricted IAM administration; +- permission to invoke unrelated Cloud Run services; +- access to unrelated buckets; +- access to every secret in the project. + +--- + +# 13. Worker Service Account + +The worker identity SHOULD receive only permissions required to: + +- connect to Cloud SQL; +- read required secrets; +- read and write configured CARE buckets; +- emit logs and metrics; +- call required email or external services. + +The worker ordinarily does not need permission to create Cloud Tasks unless +handlers enqueue follow-up tasks. + +Such permission SHALL be added only if a concrete workflow requires it. + +--- + +# 14. Jobs Service Account + +The jobs identity SHOULD receive permissions needed by administrative commands. + +These may include: + +- Cloud SQL connectivity; +- Cloud Storage access; +- secret access; +- logging; +- selected task enqueue permissions. + +Migration jobs do not automatically require full bucket access. + +Permissions SHOULD be tailored to the actual commands executed. + +--- + +# 15. Task Invoker Service Account + +The task invoker identity SHALL be used by Cloud Tasks to call the private +worker. + +It SHOULD receive: + +```text +Cloud Run invocation permission +``` + +only on the CARE worker service. + +It SHALL not receive broad API, database or bucket permissions merely because +it invokes the worker. + +The worker's own service account performs the actual work. + +--- + +# 16. Deployment Service Account + +The deployment identity MAY receive permissions to: + +- push images; +- deploy Cloud Run services; +- update jobs; +- execute migration jobs; +- manage Scheduler; +- update task queues; +- update approved secrets or secret references; +- apply Terraform in approved environments. + +Production deployment access SHOULD be limited to: + +- CI/CD; +- authorized maintainers; +- emergency operational procedures. + +--- + +# 17. Secret Management + +Secrets SHALL be stored in Secret Manager or an equivalent protected mechanism. + +Expected secrets include: + +```text +DJANGO_SECRET_KEY +database password or connection secret +SMTP credentials +JWT or JWKS private material +Sentry DSN +external API credentials +optional Redis URLs +``` + +Non-secret configuration SHOULD remain ordinary environment variables. + +Examples of non-secret values: + +```text +GCP_PROJECT_ID +GCP_REGION +CARE_TASK_BACKEND +CARE_CACHE_BACKEND +bucket names +queue names +service URLs +``` + +--- + +# 18. Secret Versioning + +Secret updates SHOULD create new versions rather than overwrite undocumented +values. + +Operators SHALL know: + +- which Cloud Run revision uses which secret reference; +- whether the service references a fixed version or latest version; +- whether a new service revision is required after rotation; +- how to roll back to a previous secret version. + +Secret rotation SHALL be tested in staging before production where practical. + +--- + +# 19. Forbidden Secret Practices + +Operators SHALL NOT: + +- commit `.env` production files; +- commit service-account JSON keys; +- paste secrets into issue trackers; +- expose secrets through Terraform outputs; +- print secrets in CI logs; +- include credentials in container images; +- store plaintext secrets in documentation; +- use patient information as secret names. + +--- + +# 20. Artifact Registry + +Artifact Registry SHALL store CARE container images. + +Images SHOULD use immutable tags: + +```text + + +``` + +Example: + +```text +care-api:3f49b8a +care-api:gcp-v2026.08.05.1 +``` + +Deployments SHOULD record the image digest. + +The same image SHOULD be used for: + +```text +API +Cloud Tasks worker +Cloud Run Jobs +optional PostgreSQL queue worker +Celery worker in compatible deployments +``` + +--- + +# 21. Image Build + +The production image build SHALL: + +1. install locked dependencies; +2. compile translations; +3. collect static files; +4. exclude development-only tools; +5. exclude local `.env` files; +6. exclude Git credentials; +7. avoid embedded secrets; +8. produce deterministic metadata; +9. identify the upstream and GCP commits; +10. pass container tests. + +A failed test SHALL prevent image promotion. + +--- + +# 22. Image Retention + +Artifact Registry retention SHOULD balance: + +- rollback requirements; +- audit requirements; +- storage cost. + +At minimum, operators SHOULD retain: + +- the current production image; +- the previous known-good image; +- recent release images; +- images referenced by active revisions; +- images required for incident investigation. + +Untagged temporary images MAY be cleaned automatically after a defined period. + +Active revision images SHALL not be deleted. + +--- + +# 23. Cloud SQL Creation + +Cloud SQL SHALL be created through Terraform. + +Initial configuration SHALL define: + +- PostgreSQL version; +- region; +- machine size; +- storage size; +- storage growth policy; +- backup schedule; +- point-in-time recovery policy; +- maintenance window; +- deletion protection; +- network access; +- database name; +- application user. + +The smallest development configuration MAY be used initially. + +Production sizing SHALL be based on measured load. + +--- + +# 24. Database Initialization + +For a new environment: + +1. create Cloud SQL; +2. create the CARE database; +3. create the CARE application user; +4. store the database credentials; +5. build and publish the application image; +6. create or update the migration job; +7. run Django migrations; +8. create cache tables if selected; +9. run required synchronization commands; +10. verify database health. + +Expected commands include: + +```bash +python manage.py migrate --noinput +python manage.py createcachetable +python manage.py sync_permissions_roles +python manage.py sync_valueset +``` + +`createcachetable` SHALL run only when the database cache backend is selected. + +--- + +# 25. Fixture Loading + +Fixtures MAY be loaded in: + +```text +development +staging +demonstration environments +``` + +Production fixture loading SHALL be intentional. + +Operators SHALL know whether fixtures: + +- create sample users; +- create sample facilities; +- create clinical sample data; +- modify value sets; +- conflict with real initialization. + +Synthetic fixtures SHALL not be mistaken for production data. + +--- + +# 26. Database Connection Budget + +Operators SHALL maintain a connection budget. + +The maximum theoretical application connections are influenced by: + +```text +API maximum instances +API workers per instance +worker maximum instances +worker process count +jobs +optional PostgreSQL queue workers +database-cache traffic +administrative connections +``` + +The configured maximum instances SHALL remain conservative until measured. + +A connection-budget document SHOULD record: + +```text +Cloud SQL connection capacity +reserved administrative headroom +API allocation +worker allocation +jobs allocation +monitoring allocation +``` + +--- + +# 27. Database Connection Exhaustion + +Signs include: + +- failed requests; +- worker task failures; +- elevated database connection count; +- connection timeout messages; +- job failures; +- readiness-check failures. + +Immediate actions MAY include: + +1. reduce Cloud Run maximum instances; +2. reduce Gunicorn worker count; +3. reduce worker concurrency; +4. inspect long-running queries; +5. inspect leaked or idle connections; +6. temporarily disable nonessential jobs; +7. increase database capacity when justified. + +Operators SHALL not solve every connection problem by increasing Cloud SQL size +without investigating application concurrency. + +--- + +# 28. Database Backups + +Production SHALL enable automated Cloud SQL backups. + +The backup policy SHALL define: + +- schedule; +- retention; +- point-in-time recovery; +- restore testing cadence; +- responsible owner; +- incident escalation. + +A backup that has never been restored in a test environment is not sufficient +evidence of recoverability. + +--- + +# 29. Restore Testing + +A restore test SHOULD periodically: + +1. create an isolated database environment; +2. restore from a selected backup or recovery point; +3. connect a non-production CARE revision; +4. run `manage.py check`; +5. verify representative records; +6. verify migrations; +7. verify application startup; +8. destroy the temporary environment after approval. + +Restore tests SHALL not expose production data to unauthorized environments. + +--- + +# 30. Database Maintenance + +Operators SHOULD monitor: + +- CPU; +- memory; +- disk utilization; +- storage growth; +- connection count; +- transaction duration; +- slow queries; +- locks; +- table growth; +- cache-table growth; +- task-state growth; +- queue-table growth if enabled. + +PostgreSQL infrastructure tables SHOULD have explicit cleanup or retention +policies. + +--- + +# 31. PostgreSQL Database Cache + +When: + +```text +CARE_CACHE_BACKEND=postgres +``` + +the configured cache table SHALL exist. + +Operators SHALL monitor: + +- table size; +- write volume; +- expired entries; +- cache query latency; +- impact on clinical queries. + +The cache table contains disposable cache values. + +Deleting all cache rows SHOULD not destroy durable CARE state. + +--- + +# 32. Cache Table Maintenance + +Database cache entries may remain until normal cache operations perform +cleanup. + +For a large deployment, operators MAY add a maintenance command or scheduled +cleanup process if measurement shows unacceptable growth. + +Any cleanup SHALL: + +- target only the cache table; +- avoid long table locks; +- be tested in staging; +- preserve active application tables; +- produce observable logs. + +The project SHALL not add complex cleanup before actual need is demonstrated. + +--- + +# 33. PostgreSQL Rate-Limit State + +If rate limiting uses explicit PostgreSQL models, operations SHALL define: + +- retention window; +- cleanup cadence; +- indexes; +- concurrency semantics; +- failure policy. + +Expired counter rows SHOULD be removed by a scheduled management command or +normal application operation. + +Security-sensitive rate limits SHALL not silently fail open without an explicit +decision. + +--- + +# 34. PostgreSQL Task Queue Operations + +This section applies only if the optional PostgreSQL queue is accepted. + +Operators SHALL monitor: + +- queued jobs; +- oldest queued job; +- active jobs; +- failed jobs; +- retry count; +- worker heartbeat; +- worker connections; +- queue-table size; +- cleanup status. + +An immediate PostgreSQL queue requires an active consumer. + +If the worker is configured with minimum instances greater than zero, its +baseline cost SHALL be included in operational budgets. + +--- + +# 35. PostgreSQL Queue Backlog + +When backlog grows: + +1. verify worker health; +2. inspect failed or locked tasks; +3. inspect database contention; +4. inspect queue concurrency; +5. inspect external dependency failures; +6. increase worker concurrency carefully; +7. increase active workers only within the database connection budget; +8. pause task producers if necessary. + +Operators SHALL not delete queued clinical work without documented review. + +--- + +# 36. Cloud Storage Buckets + +The minimum logical storage areas are: + +```text +patient +facility +report +``` + +They MAY use: + +- separate physical buckets; +- or a documented shared bucket with separate prefixes. + +Production buckets SHALL: + +- block public access; +- use IAM-based access; +- use appropriate location; +- define retention or versioning decisions; +- define lifecycle rules only when safe; +- emit access logs or audit events where required. + +--- + +# 37. Bucket Access + +The frontend SHALL not access buckets directly. + +The API and worker service accounts receive controlled access. + +Operations SHALL verify: + +- anonymous reads fail; +- anonymous writes fail; +- unrelated service accounts fail; +- authorized API and worker access succeeds; +- deletion permissions match application needs. + +--- + +# 38. Bucket Lifecycle Rules + +Lifecycle policies MAY clean: + +- abandoned temporary objects; +- old object versions; +- development test prefixes; +- expired staging data. + +Production lifecycle rules SHALL not delete active clinical objects based only +on object age unless CARE's data-retention policy explicitly requires it. + +Lifecycle configuration SHALL be reviewed as a potential data-loss mechanism. + +--- + +# 39. Bucket Versioning + +Object versioning MAY be enabled where recovery value justifies the cost. + +The decision SHALL consider: + +- accidental deletion recovery; +- accidental overwrite recovery; +- storage cost; +- retention requirements; +- deletion semantics; +- privacy erasure requirements. + +Enabling versioning does not replace database backups or application-level +authorization. + +--- + +# 40. File Upload Operations + +Uploads pass through the CARE API. + +Operators SHALL monitor: + +- request duration; +- request size; +- memory usage; +- temporary-disk usage; +- storage write failures; +- upload validation failures; +- incomplete upload records; +- API timeout rate. + +The supported maximum upload size SHALL be documented. + +Cloud Run memory and timeout configuration SHALL reflect measured upload +behavior. + +--- + +# 41. File Download Operations + +Downloads pass through CARE. + +Operators SHALL monitor: + +- streaming duration; +- first-byte latency; +- concurrent streams; +- storage read failures; +- memory use; +- API egress; +- request timeout rate. + +Large download traffic MAY materially affect Cloud Run and network cost. + +Any later move to a different download architecture SHALL require a separate +security and architecture decision. + +--- + +# 42. Missing Storage Objects + +A database record may occasionally reference a missing object because of: + +- interrupted operation; +- administrative deletion; +- permission failure; +- software defect; +- manual bucket modification. + +Operational handling SHALL include: + +- controlled API response; +- structured error log; +- object identifier; +- database record identifier; +- no full clinical payload; +- investigation or cleanup procedure. + +Operators SHALL not recreate missing clinical files with placeholder data. + +--- + +# 43. Orphaned Storage Objects + +An object may exist without a corresponding database record. + +Potential causes include: + +- database failure after upload; +- interrupted request; +- abandoned workflow; +- manual data changes. + +A cleanup process MAY identify orphaned objects. + +Automatic deletion SHALL not be implemented without a reliable way to prove +the object is unreferenced. + +A quarantine or report-only mode SHOULD precede destructive cleanup. + +--- + +# 44. Cloud Tasks Queue Operations + +The default GCP task backend uses Cloud Tasks. + +Operators SHALL monitor: + +- dispatch failures; +- queue depth; +- oldest task age; +- retry count; +- worker response codes; +- worker latency; +- task execution duration; +- dead-letter or terminal failures if configured. + +Each queue SHALL have documented: + +- purpose; +- target worker; +- retry policy; +- rate limits; +- concurrency limits; +- execution deadline; +- owner. + +--- + +# 45. Cloud Tasks Retry Policy + +Retry policy SHALL reflect task semantics. + +Email tasks may tolerate retries differently from report generation. + +Operators SHALL know: + +- maximum attempts; +- minimum backoff; +- maximum backoff; +- maximum retry duration; +- permanent failure behavior. + +Increasing retries SHALL not substitute for fixing deterministic task errors. + +--- + +# 46. Cloud Tasks Backlog + +When the queue grows unexpectedly: + +1. inspect worker health; +2. inspect Cloud Run errors; +3. inspect task response codes; +4. inspect database connectivity; +5. inspect storage and email dependencies; +6. inspect queue rate limits; +7. inspect worker maximum instances; +8. verify Cloud SQL connection headroom. + +Raising worker scale limits SHALL be coordinated with the database connection +budget. + +--- + +# 47. Failed Cloud Tasks + +A failed task investigation SHOULD capture: + +```text +task name +task identifier +attempt count +worker revision +failure category +related CARE record identifier +timestamp +``` + +It SHOULD NOT capture full sensitive payloads unless an approved diagnostic +process requires them. + +Recovery MAY involve: + +- correcting configuration; +- fixing application code; +- re-enqueueing a task; +- repairing application state; +- marking a task permanently failed. + +Re-enqueueing SHALL respect idempotency requirements. + +--- + +# 48. Private Worker Access + +The worker SHALL remain private. + +Operational verification SHALL periodically confirm: + +- unauthenticated requests fail; +- only expected invokers have access; +- the public API does not expose the internal task route unintentionally; +- worker logs identify the invoker and task; +- IAM bindings have not broadened unexpectedly. + +Application headers SHALL not replace Cloud Run IAM authentication. + +--- + +# 49. Cloud Run Jobs + +Jobs SHALL be used for: + +```text +migrations +permission synchronization +value-set synchronization +cleanup +batch operations +administrative commands +``` + +Each job SHALL define: + +- command; +- timeout; +- retries; +- service account; +- CPU and memory; +- environment variables; +- secret references; +- expected exit code; +- alerting. + +--- + +# 50. Running a Job Manually + +Operators MAY manually execute approved jobs for: + +- deployment; +- maintenance; +- incident response; +- verification. + +Before running a production job, verify: + +- correct project; +- correct region; +- correct environment; +- correct image revision; +- correct command; +- expected impact; +- concurrency safety. + +The operator SHOULD record the execution reason. + +--- + +# 51. Migration Job Operations + +Normal deployment sequence: + +1. update migration job to the new image; +2. execute the job; +3. wait for completion; +4. inspect logs; +5. stop deployment if it fails; +6. deploy worker; +7. deploy API. + +The API SHALL not deploy automatically after a failed migration. + +A successful command exit is required. + +--- + +# 52. Migration Failure + +When a migration fails: + +1. do not deploy the new API revision; +2. preserve logs; +3. determine whether the migration partially applied; +4. inspect Django migration state; +5. correct code or data in a controlled branch; +6. test on an isolated environment; +7. rerun only after review. + +Operators SHALL not blindly fake migrations in production. + +`--fake` usage requires an explicit understanding of schema state. + +--- + +# 53. Scheduled Jobs + +Cloud Scheduler SHALL trigger approved periodic work. + +Expected schedules include: + +```text +expired token-slot cleanup +incomplete file-upload cleanup +other documented maintenance commands +``` + +Each schedule SHALL define: + +- timezone; +- cadence; +- authenticated target; +- retry behavior; +- disabled or enabled state; +- owner; +- alerting. + +The configured timezone SHALL be explicit. + +--- + +# 54. Scheduler Changes + +When changing a production schedule: + +1. document the old cadence; +2. document the new cadence; +3. assess overlap; +4. assess workload duration; +5. assess database load; +6. apply through Terraform; +7. verify the next execution; +8. verify no duplicate scheduler exists. + +Celery Beat SHALL not schedule the same production job in the default GCP +profile. + +--- + +# 55. API Deployment + +The API deployment SHALL use an immutable image. + +Before deployment: + +- tests pass; +- image is published; +- migrations succeed; +- worker is deployed when required; +- configuration is validated; +- secrets exist; +- storage aliases resolve. + +After deployment: + +- health check passes; +- static file request passes; +- authenticated API smoke test passes; +- storage smoke test passes; +- task enqueue passes. + +--- + +# 56. Worker Deployment + +The worker SHOULD deploy before the API begins enqueueing tasks requiring the +new handler set. + +Deployment order: + +```text +migration job +worker +API +scheduler changes +``` + +The worker revision SHALL contain handlers required by the API revision. + +This order reduces tasks arriving before their handler exists. + +--- + +# 57. Cloud Run Revisions + +Operators SHALL preserve revision traceability. + +Each revision SHOULD identify: + +```text +image digest +Git commit +upstream commit +deployment timestamp +environment +``` + +Traffic SHOULD normally move to the newly verified revision. + +Canary or gradual traffic allocation MAY be used later if operationally useful. + +--- + +# 58. Application Rollback + +Application rollback uses a previous known-good Cloud Run revision or image. + +Before rollback, verify: + +- previous revision exists; +- secrets remain compatible; +- database schema remains compatible; +- task payloads remain compatible; +- worker and API revisions are compatible. + +API and worker may need to roll back together. + +Rollback SHALL not assume database migrations are reversible. + +--- + +# 59. Worker and API Version Compatibility + +During deployment, mixed revisions may coexist briefly. + +Task payloads SHOULD therefore remain compatible across adjacent revisions +where practical. + +Breaking task payload changes SHOULD use: + +- versioned task names; +- versioned payload schemas; +- compatible transition logic; +- controlled deployment order. + +The API SHALL not enqueue a payload unsupported by the active worker fleet. + +--- + +# 60. Health Endpoints + +The application SHOULD expose separate concepts for: + +```text +liveness +readiness +diagnostics +``` + +Liveness answers: + +```text +Is the process running? +``` + +Readiness answers: + +```text +Can the process serve its configured role? +``` + +Diagnostics may report dependency state to authorized operators. + +Sensitive infrastructure details SHALL not be publicly exposed. + +--- + +# 61. API Readiness + +API readiness SHOULD require: + +- Django startup completed; +- required settings valid; +- PostgreSQL available. + +It MAY require configured storage access if the application cannot operate +meaningfully without it. + +It SHALL not require: + +- Redis when Redis is disabled; +- Celery when Cloud Tasks is selected; +- worker availability through a synchronous request; +- every optional external integration. + +--- + +# 62. Worker Readiness + +Worker readiness SHOULD verify: + +- application startup; +- handler registry; +- PostgreSQL; +- required storage configuration; +- required secrets. + +Cloud Run IAM authorization is tested externally rather than through an +ordinary application health response. + +--- + +# 63. Logging + +All runtime processes SHALL log to stdout and stderr. + +Logs SHOULD be structured. + +Recommended fields: + +```text +severity +timestamp +environment +service +revision +request_id +task_id +task_name +task_backend +attempt +duration_ms +status +``` + +Logs SHALL avoid: + +- complete patient records; +- full request bodies; +- file contents; +- authentication tokens; +- signed credentials; +- passwords; +- secret values. + +--- + +# 64. Log Retention + +Log retention SHALL balance: + +- incident investigation; +- audit needs; +- privacy; +- cost. + +Development logs MAY use shorter retention. + +Production logs MAY require longer retention according to policy. + +Long retention SHALL not be used to justify logging sensitive payloads. + +--- + +# 65. Log-Based Alerts + +Useful alerts include: + +```text +high API 5xx rate +worker task failures +migration job failure +scheduled job failure +Cloud SQL connection exhaustion +storage permission failures +repeated authentication failures +queue backlog +optional Redis failure +``` + +Alerts SHALL include enough context to identify the environment and revision. + +They SHALL not include sensitive clinical content. + +--- + +# 66. Metrics + +Operators SHOULD monitor: + +## API + +```text +request count +latency +5xx rate +instance count +cold starts +memory +CPU +concurrency +``` + +## Worker + +```text +task count +task duration +failure rate +retry rate +instance count +cold starts +``` + +## Cloud SQL + +```text +connections +CPU +memory +storage +disk utilization +query latency +locks +``` + +## Storage + +```text +operation errors +bytes stored +request count +egress +``` + +## Tasks + +```text +queue depth +oldest task age +retry count +execution latency +``` + +--- + +# 67. Dashboards + +At minimum, dashboards SHOULD provide: + +```text +environment overview +API health +worker and task health +Cloud SQL health +storage failures +scheduled-job status +cost indicators +``` + +Operators SHOULD be able to identify: + +- which revision is failing; +- whether failures are API, database, storage or task related; +- whether the issue affects one environment or all; +- whether an upstream synchronization introduced the failure. + +--- + +# 68. Incident Severity + +The project SHOULD define severity levels. + +Example: + +## Critical + +```text +patient-data exposure +unauthorized access +database corruption +production unavailable +irrecoverable file loss +``` + +## High + +```text +major API failure +task backlog affecting clinical work +database nearing exhaustion +storage access broadly failing +``` + +## Medium + +```text +scheduled cleanup failure +optional cache failure +partial external integration outage +``` + +## Low + +```text +development environment issue +noncritical dashboard problem +documentation defect +``` + +The exact policy SHALL be adapted to the operating organization. + +--- + +# 69. Incident Response + +A production incident SHOULD follow: + +1. identify environment; +2. identify affected revision; +3. preserve relevant logs; +4. assess patient and data impact; +5. stop harmful operations if necessary; +6. roll back application code when safe; +7. disable problematic schedules or queues when necessary; +8. restore data only through approved procedures; +9. document actions; +10. perform post-incident review. + +Operators SHALL avoid destructive ad hoc commands without preserving evidence. + +--- + +# 70. Disabling a Scheduler Job + +A failing scheduled job MAY be disabled temporarily. + +The operator SHALL record: + +- job name; +- reason; +- disable time; +- expected consequence; +- owner; +- restoration condition. + +Disabling cleanup jobs may lead to: + +- stale records; +- incomplete upload accumulation; +- increased storage use. + +The operational consequence SHALL be understood. + +--- + +# 71. Pausing Task Production + +When the worker or a dependency is failing, the API MAY need to stop enqueueing +selected tasks. + +This may be implemented through: + +- configuration; +- feature flags; +- temporary endpoint restrictions; +- queue-specific controls. + +The system SHALL not silently accept an asynchronous request and discard the +work. + +The user-facing behavior SHALL be explicit. + +--- + +# 72. Optional Redis Operations + +This section applies only when Redis is enabled. + +Operators SHALL know which responsibilities use Redis: + +```text +cache +rate limiting +transient state +Celery +``` + +The application SHALL not treat Redis as one undifferentiated dependency. + +--- + +# 73. Redis Connection Configuration + +Redis connections SHOULD use: + +```text +rediss:// +``` + +when TLS is required. + +Configuration SHALL define: + +- timeout; +- connection pool behavior; +- retry policy; +- certificate verification; +- database or namespace strategy; +- maximum connections. + +For serverless providers, connection limits and request pricing SHALL be +understood. + +--- + +# 74. Upstash Operations + +When Upstash is selected, operators SHOULD monitor: + +- request volume; +- latency; +- plan limits; +- storage use; +- eviction; +- connection errors; +- regional location; +- TLS errors. + +The application SHOULD remain provider-neutral. + +Migrating away from Upstash SHOULD require configuration changes rather than +domain-code changes. + +--- + +# 75. Redis Outage + +Expected behavior depends on responsibility. + +## Performance cache + +May degrade to cache misses. + +## Rate limiting + +Must follow the documented security fallback. + +## Progress state + +May temporarily stop showing progress. + +## Celery broker + +Task dispatch fails until broker recovery. + +The outage behavior SHALL be tested before production. + +--- + +# 76. Email Operations + +CARE email tasks use Django's email abstraction. + +Operators SHALL monitor: + +- send failures; +- authentication failures; +- rate limits; +- retry volume; +- rejected recipients; +- provider outages. + +Email credentials SHALL remain in Secret Manager. + +Task logs SHOULD avoid full message contents. + +--- + +# 77. TOTP Notification Failures + +Failure to send TOTP enabled or disabled notifications SHOULD be visible. + +Operators SHALL distinguish: + +- authentication configuration error; +- transient provider error; +- invalid recipient; +- application template error; +- worker failure. + +Retries SHALL not generate uncontrolled duplicate email volume. + +--- + +# 78. Cost Monitoring + +Cost monitoring SHALL separate: + +```text +Cloud SQL +Cloud Run +Cloud Storage +Cloud Tasks +Cloud Scheduler +Artifact Registry +Secret Manager +logging +network egress +optional Redis +``` + +Labels SHOULD permit environment and component attribution. + +Development cost SHOULD not be confused with production cost. + +--- + +# 79. Persistent Baseline Cost + +The principal expected persistent baseline cost is Cloud SQL. + +Other persistent costs include: + +- stored data; +- retained backups; +- retained logs; +- retained images; +- optional managed Redis; +- a minimum-instance queue worker if selected. + +The system SHALL not be described as entirely scale-to-zero. + +--- + +# 80. Cost Controls + +Useful cost controls include: + +- Cloud Run maximum instances; +- minimum instances set to zero where appropriate; +- conservative Cloud SQL sizing; +- log retention limits; +- Artifact Registry cleanup; +- development environment shutdown or recreation; +- bucket lifecycle rules for synthetic data; +- task queue rate limits; +- alerting on unexpected spend. + +Cost controls SHALL not delete production data automatically without policy. + +--- + +# 81. Development Environment Teardown + +A development environment MAY be destroyed and recreated when it contains only +synthetic data. + +Before teardown, verify: + +- no real patient data; +- no required test evidence; +- Terraform state is correct; +- no shared production resources; +- no shared buckets or secrets. + +The teardown procedure SHALL target the correct environment explicitly. + +--- + +# 82. Staging Environment + +Staging SHOULD resemble production in: + +- service topology; +- settings profile; +- IAM pattern; +- storage backend; +- task backend; +- migration process; +- health checks; +- deployment pipeline. + +Staging MAY use smaller resources. + +It SHALL not silently use local MinIO or Celery if production uses GCS and Cloud +Tasks, unless a particular test explicitly requires that profile. + +--- + +# 83. Production Access + +Production access SHOULD use: + +- individual identities; +- multi-factor authentication; +- least privilege; +- audited role assignment; +- temporary elevation where practical. + +Shared administrator accounts SHOULD be avoided. + +Direct database access SHALL be limited. + +Administrative changes SHALL be traceable to a person or deployment identity. + +--- + +# 84. Manual Database Changes + +Manual SQL in production SHOULD be avoided. + +When required: + +1. create a reviewed script; +2. test it in staging; +3. back up the database; +4. estimate lock and runtime impact; +5. execute with a named operator; +6. record results; +7. convert recurring changes into migrations or management commands. + +Application data fixes SHALL not be hidden inside Terraform. + +--- + +# 85. Manual Bucket Changes + +Manual object deletion or movement SHOULD be avoided. + +When required: + +- verify the corresponding database record; +- verify authorization; +- preserve evidence; +- consider versioning; +- document the action; +- verify application behavior afterward. + +Objects SHALL not be renamed manually unless database references are updated +consistently. + +--- + +# 86. Key Rotation + +Credential rotation procedures SHOULD exist for: + +```text +database password +SMTP credentials +JWT or JWKS material +optional Redis credentials +external API credentials +``` + +Rotation steps SHOULD include: + +1. create new credential; +2. add new secret version; +3. deploy or refresh dependent service; +4. verify operation; +5. revoke old credential; +6. verify no old revision still requires it; +7. document completion. + +--- + +# 87. Django Secret Key Rotation + +Rotating `DJANGO_SECRET_KEY` may invalidate: + +- signed cookies; +- sessions; +- tokens or signatures depending on application use. + +Rotation SHALL be planned. + +If Django supports fallback signing keys in the deployed version, they MAY be +used during a controlled transition. + +The exact procedure SHALL be validated against the CARE version. + +--- + +# 88. Service Account Key Policy + +Cloud Run runtime identities SHOULD use attached service accounts. + +Static service-account key files SHOULD not be created for ordinary operation. + +If an exceptional integration requires a key: + +- justify it; +- restrict permissions; +- store it in Secret Manager; +- rotate it; +- monitor use; +- define removal plans. + +--- + +# 89. Release Procedure + +A production release SHOULD follow: + +1. synchronize with upstream as planned; +2. merge approved feature work into `gcp`; +3. run CI; +4. build immutable image; +5. deploy to staging; +6. run staging smoke tests; +7. approve production plan; +8. run production migrations; +9. deploy worker; +10. deploy API; +11. update schedules; +12. run production smoke tests; +13. tag the release; +14. record upstream and image references. + +--- + +# 90. Release Metadata + +Each release SHOULD record: + +```text +release tag +GCP commit SHA +upstream commit SHA +image digest +Terraform commit SHA +database migration state +deployment timestamp +operator or pipeline +``` + +This metadata MAY be stored in: + +- release notes; +- deployment annotations; +- a release record; +- `UPSTREAM_BASE`; +- application version endpoint. + +--- + +# 91. First Production Release + +Because the deployment is greenfield, the first production release SHALL +verify: + +- empty database initialization; +- first administrator workflow; +- first facility workflow; +- first user workflow; +- first patient workflow; +- first upload and download; +- first report; +- first task; +- first scheduled job; +- first backup; +- initial monitoring and alerting. + +The environment SHALL not be considered ready merely because the API health +endpoint returns success. + +--- + +# 92. Routine Maintenance + +Routine operational work includes: + +```text +upstream synchronization +dependency updates +security updates +database backup verification +restore testing +log-retention review +cost review +IAM review +secret rotation +task backlog review +scheduled-job review +storage-growth review +container cleanup +``` + +A maintenance calendar SHOULD assign cadence and ownership. + +--- + +# 93. Suggested Maintenance Cadence + +Example cadence: + +## Daily + +```text +critical alerts +failed jobs +task backlog +API availability +database capacity warnings +``` + +## Weekly + +```text +error trends +scheduler success +storage failures +unexpected cost changes +``` + +## Monthly + +```text +upstream changes +dependency updates +IAM review +Artifact Registry cleanup +cache and queue table growth +backup status +``` + +## Quarterly + +```text +restore test +secret rotation review +disaster-recovery review +production access review +capacity review +``` + +The operating organization MAY adjust this cadence. + +--- + +# 94. Dependency Updates + +Dependency updates SHALL: + +- use the repository's package manager; +- update lockfiles; +- run local tests; +- run storage tests; +- run task tests; +- build the production image; +- validate GCP settings; +- run staging smoke tests for significant updates. + +Special attention SHALL be paid to: + +```text +Django +django-storages +Google client libraries +Celery +Redis client +PostgreSQL queue library if enabled +Gunicorn +``` + +--- + +# 95. Security Updates + +Security updates affecting deployed components SHALL be prioritized. + +The process SHOULD: + +1. identify exposure; +2. update the affected dependency or code; +3. run focused tests; +4. deploy to staging; +5. deploy an immutable production revision; +6. record the fix; +7. verify no vulnerable revision receives traffic. + +The current image digest SHALL be known during incident response. + +--- + +# 96. Disaster Recovery Scope + +Disaster recovery SHALL address at least: + +```text +Cloud SQL loss or corruption +bucket deletion or object loss +incorrect application deployment +secret compromise +service-account compromise +region-level service disruption +Terraform state loss +``` + +Not every scenario requires active multi-region deployment. + +The expected recovery time and recovery point SHALL be documented according to +actual organizational needs. + +--- + +# 97. Cloud SQL Recovery + +Recovery options MAY include: + +- point-in-time recovery; +- backup restore; +- restore to a new instance; +- application reconnection; +- DNS or configuration update; +- deployment of a compatible CARE revision. + +Recovery SHALL be tested. + +Operators SHALL not overwrite the failed source before investigation unless +urgently required. + +--- + +# 98. Cloud Storage Recovery + +Recovery depends on configured protections: + +- versioning; +- retention; +- backups or exports; +- application metadata. + +A database restore alone may not restore deleted objects. + +A bucket recovery plan SHALL account for: + +- object names; +- database references; +- versions; +- permissions; +- lifecycle policies. + +--- + +# 99. Terraform State Recovery + +Terraform state recovery SHALL use the remote backend's version history or +backup process. + +Before applying with reconstructed state: + +- verify resource identities; +- avoid accidental recreation; +- import existing resources when necessary; +- review the plan carefully; +- protect Cloud SQL and buckets. + +Loss of state SHALL not trigger immediate destruction and recreation of +production. + +--- + +# 100. Compromised Secret Response + +When a secret is suspected compromised: + +1. identify affected services; +2. create a replacement credential; +3. store a new secret version; +4. deploy dependent services; +5. revoke the old credential; +6. inspect access logs; +7. assess patient-data impact; +8. document the incident. + +Rotating the secret without investigating use may be insufficient. + +--- + +# 101. Compromised Service Account + +Response SHOULD include: + +1. disable or restrict the identity; +2. inspect IAM bindings; +3. inspect audit logs; +4. replace affected deployment identity; +5. rotate any associated keys; +6. redeploy services if necessary; +7. investigate accessed resources; +8. document scope and impact. + +Runtime service accounts SHOULD not use static keys, reducing this risk. + +--- + +# 102. Operational Runbook Format + +Individual operational procedures SHOULD use a consistent template: + +```markdown +# Procedure name + +## Purpose + +## Preconditions + +## Environment + +## Impact + +## Commands or actions + +## Verification + +## Rollback + +## Logging and evidence + +## Owner +``` + +High-risk actions SHALL not exist only as informal chat instructions. + +--- + +# 103. Required Runbooks + +Before production use, create runbooks for: + +```text +deploy CARE +roll back API and worker +run database migrations +restore Cloud SQL backup +rotate database password +rotate Django secret +disable a Scheduler job +re-enqueue a failed task +investigate task backlog +investigate missing object +investigate database connection exhaustion +recreate development environment +synchronize upstream +``` + +--- + +# 104. Operational Documentation Storage + +Operational documentation SHALL be stored in the repository when it contains no +secrets. + +Suggested layout: + +```text +docs/xii/architecture/runbooks/ +├── deploy.md +├── rollback.md +├── migrate.md +├── restore-database.md +├── rotate-secrets.md +├── task-backlog.md +├── missing-file.md +├── scheduler.md +└── recreate-dev.md +``` + +Secret values and confidential incident details SHALL remain outside Git. + +--- + +# 105. Production Readiness Checklist + +Before first production use: + +- [ ] Terraform state is remote and protected. +- [ ] Cloud SQL backups are enabled. +- [ ] Point-in-time recovery decision is documented. +- [ ] Cloud Storage buckets are private. +- [ ] Service accounts follow least privilege. +- [ ] No runtime service-account keys are used. +- [ ] Secrets are in Secret Manager. +- [ ] API runs on Cloud Run. +- [ ] Worker is private. +- [ ] Cloud Tasks invocation succeeds. +- [ ] Cloud Scheduler jobs are correct. +- [ ] Migration job succeeds. +- [ ] Storage upload and download pass through CARE. +- [ ] PostgreSQL cache works when selected. +- [ ] Redis is not required by the default profile. +- [ ] Logs exclude sensitive payloads. +- [ ] Alerts are configured. +- [ ] Restore procedure is documented. +- [ ] Application rollback is tested. +- [ ] Upstream base is recorded. +- [ ] Synthetic end-to-end tests pass. +- [ ] Initial administrator procedure is documented. + +--- + +# 106. Routine Deployment Checklist + +Before deployment: + +- [ ] Correct environment selected. +- [ ] Worktree and branch verified. +- [ ] CI passed. +- [ ] Image digest recorded. +- [ ] Terraform plan reviewed. +- [ ] Migrations reviewed. +- [ ] Worker/API compatibility verified. +- [ ] Secrets exist. +- [ ] Backup status verified for high-risk migrations. + +During deployment: + +- [ ] Migration job succeeded. +- [ ] Worker deployed. +- [ ] API deployed. +- [ ] Scheduler changes applied. +- [ ] Smoke tests passed. + +After deployment: + +- [ ] Error rate normal. +- [ ] Task failures normal. +- [ ] Database connections normal. +- [ ] Storage operations normal. +- [ ] Release metadata recorded. + +--- + +# 107. Operational Definition of Healthy + +The production environment is healthy when: + +- API readiness passes; +- authenticated CARE workflows succeed; +- database connections remain within budget; +- task queue age remains acceptable; +- worker failures remain within expected levels; +- scheduled jobs succeed; +- storage reads and writes succeed; +- backups complete; +- no critical security alerts exist; +- costs remain within expected range. + +A green process-health indicator alone is not sufficient. + +--- + +# 108. Operational Definition of Degraded + +The environment is degraded when one or more non-total failures exist, such as: + +```text +email provider unavailable +report generation delayed +optional Redis cache unavailable +scheduled cleanup failing +task retries elevated +storage latency elevated +``` + +The application MAY remain usable. + +Operators SHALL communicate the affected capability and expected impact. + +--- + +# 109. Operational Definition of Unavailable + +The environment is unavailable when core CARE use cannot proceed, such as: + +```text +API inaccessible +Cloud SQL unavailable +authentication broadly failing +authorization broadly failing +required storage inaccessible +critical data corruption +``` + +Unavailable status requires immediate incident handling. + +--- + +# 110. Definition of Operational Readiness + +CARE is operationally ready when: + +- infrastructure can be created reproducibly; +- the application can be deployed from an immutable image; +- the database initializes from migrations; +- secrets are managed securely; +- API and worker roles are observable; +- files are stored and served through Django; +- tasks are visible and recoverable; +- scheduled jobs are visible; +- PostgreSQL cache is maintainable; +- optional Redis behavior is documented; +- backups and restore are tested; +- release and rollback procedures exist; +- cost and capacity are monitored; +- upstream synchronization is operationalized. + +--- + +## 111. Next Document + +The next document is: + +```text +docs/xii/architecture/07-configuration-reference.md +``` + +It will define: + +- required environment variables; +- optional environment variables; +- backend-selection values; +- Cloud Run configuration; +- Cloud SQL configuration; +- Django Storage aliases; +- Cloud Tasks configuration; +- PostgreSQL cache configuration; +- optional PostgreSQL queue configuration; +- optional Redis configuration; +- validation rules; +- safe defaults; +- prohibited production defaults. diff --git a/docs/xii/architecture/07-configuration-reference.md b/docs/xii/architecture/07-configuration-reference.md new file mode 100644 index 0000000000..b2133e3c0d --- /dev/null +++ b/docs/xii/architecture/07-configuration-reference.md @@ -0,0 +1,2879 @@ +--- +title: GCP Configuration Reference +document: 07-configuration-reference +version: 0.1.0 +status: Draft +source_repository: https://github.com/ohcnetwork/care +target_platform: Google Cloud Platform +deployment_type: Greenfield +depends_on: + - docs/xii/architecture/00-scope-and-goals.md + - docs/xii/architecture/01-current-runtime.md + - docs/xii/architecture/02-target-runtime.md + - docs/xii/architecture/03-migration-plan.md + - docs/xii/architecture/04-testing.md + - docs/xii/architecture/05-upstream-sync.md + - docs/xii/architecture/06-operations.md +--- + +# GCP Configuration Reference + +## 1. Purpose + +This document defines the configuration contract for the greenfield CARE +deployment on Google Cloud Platform. + +It specifies: + +- environment variables; +- supported backend values; +- required and optional settings; +- validation rules; +- safe defaults; +- production restrictions; +- compatibility variables; +- role-specific configuration; +- backend-specific configuration. + +This document describes the intended configuration interface. + +Exact implementation details MAY differ where required by the current CARE, +Django, `django-storages` or Google Cloud library versions. + +Any implementation difference SHALL preserve the behavior described here. + +--- + +## 2. Configuration Principles + +CARE configuration SHALL follow these principles. + +### 2.1 Environment-based configuration + +Deployment configuration SHALL be injected through: + +- environment variables; +- Secret Manager references; +- Cloud Run service configuration; +- Cloud Run Job configuration; +- Terraform variables. + +Production configuration SHALL NOT be stored in committed `.env` files. + +### 2.2 Explicit backend selection + +Backend choices SHALL use explicit variables. + +Examples: + +```text +CARE_TASK_BACKEND=cloud_tasks +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +``` + +The application SHALL NOT infer the entire runtime from variables such as: + +```text +IS_GCP=true +USE_SERVERLESS=true +PRODUCTION_PROVIDER=google +``` + +### 2.3 Validate only selected backends + +Variables required by an unused backend SHALL not be mandatory. + +For example: + +```text +REDIS_CACHE_URL +``` + +SHALL not be required when: + +```text +CARE_CACHE_BACKEND=postgres +``` + +Likewise: + +```text +GCP_TASKS_QUEUE +``` + +SHALL not be required when: + +```text +CARE_TASK_BACKEND=celery +``` + +### 2.4 Fail clearly + +Invalid or incomplete production configuration SHALL fail during startup or +deployment validation with a clear error. + +Errors SHOULD identify: + +- variable name; +- invalid or missing value; +- selected backend; +- supported alternatives. + +### 2.5 No secret aliases with insecure defaults + +Production secrets SHALL not have usable development defaults. + +The GCP settings SHALL not silently use values such as: + +```text +secret +changeme +postgres +minioadmin +``` + +for production credentials. + +--- + +# 3. Configuration Layers + +The configuration is divided into these groups: + +```text +Django core +application identity +Cloud Run +database +storage +task execution +cache +rate limiting +transient state +optional Redis +email +security +logging and monitoring +service-specific roles +jobs and schedules +``` + +--- + +# 4. Settings Module + +## 4.1 `DJANGO_SETTINGS_MODULE` + +Required for all GCP roles. + +```text +DJANGO_SETTINGS_MODULE=config.settings.deployment +``` + +`config.settings.deployment` is the production settings module the repository +already ships; GCP introduces no settings module of its own. Provider selection +is configuration, not code (ADR-0001), so the same module serves AWS and GCP. + +The API, worker and jobs SHOULD all use the same settings module unless a +future role requires a narrowly scoped alternative. + +Production SHALL NOT use: + +```text +config.settings.local +``` + +The Celery compatibility runtime MAY use an existing production or +deployment-oriented settings module outside GCP. + +--- + +# 5. Environment Identity + +## 5.1 `CARE_ENVIRONMENT` + +Required. + +Supported values: + +```text +dev +staging +prod +``` + +Example: + +```text +CARE_ENVIRONMENT=prod +``` + +This value SHOULD be included in: + +- logs; +- Sentry environment; +- metrics labels; +- deployment annotations; +- task metadata. + +Unknown values SHALL be rejected unless explicitly allowed for ephemeral test +environments. + +## 5.2 `APP_VERSION` + +Recommended. + +Example: + +```text +APP_VERSION=gcp-v2026.08.05.1 +``` + +This value identifies the logical application release. + +## 5.3 `GIT_COMMIT_SHA` + +Recommended. + +Example: + +```text +GIT_COMMIT_SHA=3f49b8a... +``` + +It SHOULD match the source used to build the deployed image. + +## 5.4 `UPSTREAM_COMMIT_SHA` + +Recommended. + +This records the upstream CARE commit included in the fork. + +## 5.5 `DEPLOYED_AT` + +Optional. + +ISO-8601 timestamp identifying deployment time. + +--- + +# 6. Django Core Configuration + +## 6.1 `DJANGO_SECRET_KEY` + +Required secret. + +Example reference: + +```text +projects//secrets/care-django-secret/versions/ +``` + +It SHALL: + +- contain sufficient entropy; +- remain secret; +- differ between environments; +- not use a development default; +- be rotated only through a documented procedure. + +## 6.2 `DJANGO_DEBUG` + +Production-required value: + +```text +DJANGO_DEBUG=false +``` + +`true` SHALL be prohibited in production. + +Development and controlled staging environments MAY enable debugging only when +access is restricted and sensitive data is absent. + +## 6.3 `DJANGO_ALLOWED_HOSTS` + +Required. + +Recommended representation: + +```text +DJANGO_ALLOWED_HOSTS=["care-api.example.org",".run.app"] +``` + +The exact parser MAY accept JSON or comma-separated values according to the +existing CARE environment helper. + +A wildcard: + +```text +* +``` + +SHALL not be the normal production value. + +## 6.4 `CSRF_TRUSTED_ORIGINS` + +Required for browser-facing deployments. + +Example: + +```text +CSRF_TRUSTED_ORIGINS=[ + "https://care.example.org", + "https://api.care.example.org" +] +``` + +Origins SHALL include schemes. + +## 6.5 `CORS_ALLOWED_ORIGINS` + +Required when the frontend uses a separate origin. + +Example: + +```text +CORS_ALLOWED_ORIGINS=[ + "https://care.example.org" +] +``` + +Production SHALL not enable unrestricted CORS. + +## 6.6 `CORS_ALLOWED_ORIGIN_REGEXES` + +Optional. + +Use only when exact origin lists are insufficient. + +Regular expressions SHALL be reviewed to avoid broad origin access. + +## 6.7 `DJANGO_SECURE_SSL_REDIRECT` + +Recommended production value: + +```text +true +``` + +Cloud Run proxy handling SHALL correctly recognize forwarded HTTPS. + +## 6.8 `DJANGO_SECURE_HSTS_INCLUDE_SUBDOMAINS` + +Recommended after the domain strategy is confirmed. + +## 6.9 `DJANGO_SECURE_HSTS_PRELOAD` + +SHALL not be enabled casually. + +Preload has operational consequences outside CARE. + +## 6.10 `DJANGO_SECURE_CONTENT_TYPE_NOSNIFF` + +Recommended value: + +```text +true +``` + +--- + +# 7. Cloud Run Configuration + +## 7.1 `PORT` + +Provided automatically by Cloud Run. + +The API and HTTP worker SHALL listen on: + +```text +0.0.0.0:${PORT} +``` + +The application SHALL not require a fixed production port. + +## 7.2 `CARE_PROCESS_ROLE` + +Recommended. + +Supported values: + +```text +api +task_worker +postgres_queue_worker +job +celery_worker +``` + +Example: + +```text +CARE_PROCESS_ROLE=api +``` + +The role MAY control: + +- startup command; +- route availability; +- health checks; +- enabled integrations; +- logging metadata. + +It SHALL not alter clinical business behavior. + +## 7.3 `GUNICORN_WORKERS` + +Optional. + +Conservative initial value: + +```text +GUNICORN_WORKERS=1 +``` + +The value SHALL be selected together with: + +- Cloud Run concurrency; +- memory; +- database connections; +- thread count. + +## 7.4 `GUNICORN_THREADS` + +Optional. + +Example: + +```text +GUNICORN_THREADS=4 +``` + +The total request concurrency per instance SHALL remain compatible with +database and file-streaming behavior. + +## 7.5 `GUNICORN_TIMEOUT` + +Optional. + +The value SHALL support expected API and file-transfer durations. + +It SHALL not be increased indefinitely to hide inefficient or inappropriate +request workloads. + +## 7.6 `GUNICORN_GRACEFUL_TIMEOUT` + +Optional. + +Should allow orderly request completion during revision replacement. + +## 7.7 `GUNICORN_KEEPALIVE` + +Optional. + +Use a conservative value compatible with Cloud Run proxy behavior. + +## 7.8 Cloud Run minimum instances + +Managed through Terraform rather than Django environment variables. + +Recommended default: + +```text +API: 0 +HTTP task worker: 0 +``` + +A minimum greater than zero requires explicit cost justification. + +## 7.9 Cloud Run maximum instances + +Managed through Terraform. + +The value SHALL respect the database connection budget. + +--- + +# 8. Google Cloud Identity + +## 8.1 `GCP_PROJECT_ID` + +Required for GCP environments. + +Example: + +```text +GCP_PROJECT_ID=care-production +``` + +## 8.2 `GCP_REGION` + +Required. + +Example: + +```text +GCP_REGION=us-central1 +``` + +The selected region SHOULD align with: + +- Cloud Run; +- Cloud SQL; +- Cloud Tasks location; +- storage location where practical; +- expected users; +- legal and organizational requirements. + +## 8.3 `GOOGLE_APPLICATION_CREDENTIALS` + +SHALL normally be absent in Cloud Run. + +Cloud Run SHALL use its attached service account and Application Default +Credentials. + +This variable MAY be used locally for controlled integration tests. + +Committed credential files are prohibited. + +--- + +# 9. Database Configuration + +## 9.1 `DATABASE_URL` + +Required secret or protected configuration. + +Example conceptual value: + +```text +postgresql://care:@/?host=/cloudsql/ +``` + +The exact form SHALL follow the chosen Cloud SQL connection mechanism and the +existing CARE environment parser. + +The URL SHALL not be logged. + +## 9.2 `CONN_MAX_AGE` + +Required to be deliberately configured. + +Conservative initial example: + +```text +CONN_MAX_AGE=60 +``` + +The final value SHALL be determined through testing. + +A longer lifetime may reduce connection setup overhead but retain more +connections. + +## 9.3 `CONN_HEALTH_CHECKS` + +Recommended where supported. + +Example: + +```text +CONN_HEALTH_CHECKS=true +``` + +## 9.4 `CARE_DATABASE_APPLICATION_NAME` + +Optional. + +May identify application role in PostgreSQL sessions. + +Examples: + +```text +care-api +care-worker +care-jobs +``` + +## 9.5 `CARE_DATABASE_STATEMENT_TIMEOUT_MS` + +Optional. + +A deployment MAY configure a statement timeout. + +It SHALL be tested against report generation, cleanup and administrative +commands. + +## 9.6 `CARE_DATABASE_LOCK_TIMEOUT_MS` + +Optional. + +Useful to prevent indefinitely waiting on locks. + +The value SHALL not cause normal migrations or transactions to fail +unnecessarily. + +--- + +# 10. Database Initialization Configuration + +## 10.1 `CARE_CREATE_CACHE_TABLE` + +Optional deployment-pipeline variable. + +Example: + +```text +CARE_CREATE_CACHE_TABLE=true +``` + +It indicates whether the initialization pipeline should run: + +```bash +python manage.py createcachetable +``` + +The application runtime SHALL not create tables automatically on every startup. + +## 10.2 `CARE_RUN_SYNC_PERMISSIONS` + +Optional job control. + +Example: + +```text +CARE_RUN_SYNC_PERMISSIONS=true +``` + +## 10.3 `CARE_RUN_SYNC_VALUESETS` + +Optional job control. + +Example: + +```text +CARE_RUN_SYNC_VALUESETS=true +``` + +These variables belong to deployment or job orchestration rather than +request-time application behavior. + +--- + +# 11. Storage Backend Selection + +## 11.1 General rule + +Storage provider selection SHALL occur through Django `STORAGES`. + +Application code SHALL use logical aliases. + +The application SHALL not use: + +```text +CARE_STORAGE_PROVIDER=gcp +``` + +inside business logic to branch between SDKs. + +The settings module MAY use provider selection to construct `STORAGES`. + +## 11.2 `CARE_STORAGE_BACKEND` + +**Implemented in IS-01** (`config/storage.py`, `config/settings/base.py`). + +Supported values: + +```text +s3 +gcs +``` + +Intended use: + +```text +s3 -> MinIO, AWS S3 or compatible service (default) +gcs -> GCP production +``` + +Default: + +```text +CARE_STORAGE_BACKEND=s3 +``` + +The default preserves the existing local MinIO behaviour, so no local +configuration change is required. + +Production GCP value: + +```text +CARE_STORAGE_BACKEND=gcs +``` + +An unsupported value raises `ImproperlyConfigured` at startup, naming the +supported values. + +`filesystem` is **not** a supported value. Per ES-01 §9 a filesystem backend may +remain test-only and is not exposed as a production option; tests substitute +`django.core.files.storage.InMemoryStorage` through `override_settings` instead. + +## 11.3 `CARE_PATIENT_STORAGE_ALIAS` + +Optional. + +Default: + +```text +patient +``` + +Changing logical aliases is discouraged. + +## 11.4 `CARE_FACILITY_STORAGE_ALIAS` + +Optional. + +Default: + +```text +facility +``` + +## 11.5 `CARE_REPORT_STORAGE_ALIAS` + +Optional. + +Default: + +```text +report +``` + +--- + +# 12. GCS Storage Configuration + +## 12.1 `CARE_PATIENT_STORAGE_BUCKET` + +Required when GCS is selected. + +## 12.2 `CARE_FACILITY_STORAGE_BUCKET` + +Required when GCS is selected. + +## 12.3 `CARE_REPORT_STORAGE_BUCKET` + +Required when GCS is selected. + +The report bucket MAY equal the patient bucket. + +Logical aliases SHALL remain separate. + +## 12.4 `GCS_PROJECT_ID` + +Optional alias. + +The implementation SHOULD normally reuse: + +```text +GCP_PROJECT_ID +``` + +A separate project value MAY be supported for cross-project buckets only if +required. + +## 12.5 `GCS_LOCATION` + +Infrastructure-level value. + +Managed by Terraform when creating buckets. + +It is not normally needed by Django after bucket creation. + +## 12.6 `GCS_DEFAULT_ACL` + +Production SHOULD not depend on public or object-level default ACLs. + +Uniform bucket-level access is preferred. + +## 12.7 `GCS_QUERYSTRING_AUTH` + +Recommended target value: + +```text +false +``` + +because the normal file flow passes through Django and does not expose signed +provider URLs. + +## 12.8 `GCS_FILE_OVERWRITE` + +The value SHALL reflect CARE's object-name policy. + +**Resolved in IS-01: overwrite SHALL be enabled, on every object-storage alias +and on both backends.** `config/storage.py` sets `file_overwrite: True` +unconditionally; it is not driven by an environment variable. + +The earlier suggestion that `false` "may be appropriate" for unique, immutable +names is **incorrect**, and tested to be so. `file_overwrite = False` does not +reject a duplicate name — Django's `Storage.get_available_name` silently *renames* +the object, returning e.g. `patient/_a1b2c3`. CARE derives the key +from `internal_name` on every subsequent read, so the rename is never recorded +and the database row would point at an object that does not exist. That is +precisely the "duplicate-name handling affects database object references" +hazard this section warns about, and `false` causes it rather than preventing it. + +`True` also matches the behaviour being replaced: `boto3.put_object` overwrote +unconditionally. + +In practice collisions do not occur — `internal_name` is a UUID plus a timestamp +— so the setting matters only as a guarantee. + +Verified by: + +- `care/utils/tests/test_storage_config.py` — every alias is built with + `file_overwrite: True`; +- `care/emr/tests/test_storage.py` — against real MinIO, re-saving returns the + same name and replaces the content; +- the same file demonstrates the failure mode: `InMemoryStorage`, which has no + such option, renames on collision. + +## 12.9 `GCS_MAX_MEMORY_SIZE` + +Optional backend setting. + +This value SHALL be coordinated with Django upload handlers and Cloud Run +memory. + +It SHALL not cause large objects to be loaded fully into memory. + +--- + +# 13. S3 and MinIO Configuration + +These settings apply when: + +```text +CARE_STORAGE_BACKEND=s3 +``` + +**Corrected 2026-08-07 to match the implementation.** Earlier revisions of this +section specified `S3_ACCESS_KEY`, `S3_SECRET_KEY`, `S3_ENDPOINT_URL`, +`S3_REGION_NAME`, `S3_ADDRESSING_STYLE` and `S3_SIGNATURE_VERSION`. **No such +settings exist.** ES-01 §11.1 required reusing the tracked local configuration +rather than inventing a parallel set, so the credential and endpoint variables +below are the pre-existing ones. + +## 13.1 Credentials and endpoints, per alias + +Each alias draws from its own set. All have defaults, so a local checkout needs +no new value. + +| Alias | Region | Key | Secret | Endpoint | +| --- | --- | --- | --- | --- | +| `patient` | `FILE_UPLOAD_REGION` | `FILE_UPLOAD_KEY` | `FILE_UPLOAD_SECRET` | `FILE_UPLOAD_BUCKET_ENDPOINT` | +| `report` | `FILE_UPLOAD_REGION` | `FILE_UPLOAD_KEY` | `FILE_UPLOAD_SECRET` | `FILE_UPLOAD_BUCKET_ENDPOINT` | +| `facility` | `FACILITY_S3_REGION_CODE` | `FACILITY_S3_KEY` | `FACILITY_S3_SECRET` | `FACILITY_S3_BUCKET_ENDPOINT` | + +Each falls back to the shared `BUCKET_REGION` / `BUCKET_KEY` / `BUCKET_SECRET` / +`BUCKET_ENDPOINT`. An endpoint is emitted only when set, so AWS S3 works without +one; MinIO sets `BUCKET_ENDPOINT=http://minio:9000`. + +## 13.2 `BUCKET_PROVIDER` + +Credential source, not provider selection — provider selection is +`CARE_STORAGE_BACKEND` alone. + +```text +AWS_ROLE_BASED -> omit key, secret and endpoint; the SDK resolves the + instance role +anything else -> supply key and secret explicitly +``` + +## 13.3 Bucket variables + +```text +CARE_PATIENT_STORAGE_BUCKET +CARE_FACILITY_STORAGE_BUCKET +CARE_REPORT_STORAGE_BUCKET +``` + +Defaulting to `FILE_UPLOAD_BUCKET`, `FACILITY_S3_BUCKET` and +`FILE_UPLOAD_BUCKET` respectively. These are the **only** place a bucket name is +resolved; nothing else derives one. + +## 13.4 Not configurable + +`addressing_style` and `signature_version` are **not** exposed. botocore's +defaults are used, which are correct for both AWS S3 and MinIO (verified against +the local MinIO container). + +**Known limitation:** an S3-compatible provider that requires explicit path-style +addressing or `s3v4` signing cannot currently be configured. No such provider is +in use. Adding them is a small change to `build_object_storage` in +`config/storage.py` if one appears. + +--- + +# 14. Static File Configuration + +## 14.1 `STATIC_URL` + +Existing CARE value MAY remain. + +## 14.2 `STATIC_ROOT` + +Set in Django settings or image build configuration. + +## 14.3 Static backend + +The target backend remains: + +```text +whitenoise.storage.CompressedManifestStaticFilesStorage +``` + +No runtime variable is required unless the project intentionally makes it +configurable. + +## 14.4 `CARE_COLLECTSTATIC` + +Build-time or deployment variable only. + +The production image SHOULD run `collectstatic` during build. + +The API SHALL not run it during every instance startup. + +--- + +# 15. Upload Configuration + +**Corrected 2026-08-07 to match the implementation.** Earlier revisions +specified `CARE_MAX_UPLOAD_SIZE`, `CARE_ALLOWED_UPLOAD_MIME_TYPES`, +`CARE_ALLOWED_UPLOAD_EXTENSIONS` and `CARE_BLOCKED_UPLOAD_EXTENSIONS`. **None +exists.** ES-02 §12 requires reusing CARE's existing limit rather than inventing +one, so the settings below are the real ones. + +Uploads use `multipart/form-data` (ADR-0002). See the frontend file-flow +inventory §12 for the request contract. + +## 15.1 `MAX_FILE_UPLOAD_SIZE` + +The maximum accepted upload, **in megabytes**. + +```text +MAX_FILE_UPLOAD_SIZE=5 +``` + +Defined at `config/settings/config.py`. CARE compares it against +`UploadedFile.size` before anything is written, so an oversized file is rejected +without touching storage. + +Note the unit: this is MB, not bytes, unlike the two Django settings below. + +## 15.2 `FILE_UPLOAD_MAX_MEMORY_SIZE` + +Bytes. Above this, Django's upload handlers spool the upload to a +`TemporaryUploadedFile` instead of holding it in memory. + +```text +FILE_UPLOAD_MAX_MEMORY_SIZE=2621440 +``` + +Django's own default, stated explicitly by ES-02 so the limit is visible rather +than implicit. At the defaults, anything over 2.5 MB is temp-file backed while +`MAX_FILE_UPLOAD_SIZE` still caps the total at 5 MB. + +## 15.3 `DATA_UPLOAD_MAX_MEMORY_SIZE` + +Bytes. Bounds the **non-file** part of a request body. + +```text +DATA_UPLOAD_MAX_MEMORY_SIZE=2621440 +``` + +Multipart file parts are exempt, so this does **not** cap upload size. It did +cap the previous base64 transport, where the file travelled as JSON: a 5 MB file +became roughly 6.7 MB of body and exceeded this limit. Multipart removes that +interaction. + +## 15.4 `FILE_UPLOAD_TEMP_DIR` + +Not configured. Django's default temporary directory is used. + +Set it only if a deployment needs temporary uploads on a specific volume. +Temporary files are ephemeral and SHALL NOT be treated as durable storage. + +## 15.5 `ALLOWED_MIME_TYPES` + +The MIME allowlist, defined at `config/settings/base.py`. + +The value checked against it is **sniffed from the file's leading bytes** with +`python-magic`, not taken from the request. A browser-declared `Content-Type` is +never trusted. + +## 15.6 Extension rules + +Not settings. Extension policy lives in `FileNameValidator` +(`care/utils/models/validators.py`), applied through `FileUploadCreateSpec`. +Security-sensitive defaults are in code, not deployment configuration. + +--- + +# 16. Download Configuration + +## 16.1 Inline MIME types + +**Not a setting.** The inline allowlist is `SAFE_INLINE_FORMATS` in +`care/emr/utils/file_download.py`. Types in it are served with: + +```text +Content-Disposition: inline +``` + +and everything else as `attachment`. This preserves the behaviour the presigned +`ResponseContentDisposition` used to provide. ES-02 §21 forbids broadening it +without a verified requirement. + +Likely candidates: + +```text +application/pdf +selected image types +``` + +## 16.2 `CARE_DOWNLOAD_CHUNK_SIZE` + +Optional. + +May control streaming chunk size where the implementation exposes it. + +The value SHALL be benchmarked. + +## 16.3 `CARE_ENABLE_RANGE_REQUESTS` + +Optional. + +Default: + +```text +false +``` + +unless media-range support is implemented and tested. + +--- + +# 17. Task Backend Selection + +## 17.1 `CARE_TASK_BACKEND` + +Required. + +Supported values: + +```text +cloud_tasks +celery +postgres +``` + +The `postgres` value SHALL be accepted only if the PostgreSQL queue backend is +implemented and approved. + +Recommended GCP value: + +```text +CARE_TASK_BACKEND=cloud_tasks +``` + +Local upstream-compatible value: + +```text +CARE_TASK_BACKEND=celery +``` + +## 17.2 `CARE_TASK_DEFAULT_DELAY_SECONDS` + +Optional. + +Default: + +```text +0 +``` + +## 17.3 `CARE_TASK_PAYLOAD_VERSION` + +Optional. + +Recommended when task payload schemas become versioned. + +Example: + +```text +CARE_TASK_PAYLOAD_VERSION=1 +``` + +## 17.4 `CARE_TASK_MAX_PAYLOAD_BYTES` + +Recommended. + +Prevents oversized task payloads. + +Tasks SHOULD normally contain identifiers rather than clinical records. + +--- + +# 18. Cloud Tasks Configuration + +Required when: + +```text +CARE_TASK_BACKEND=cloud_tasks +``` + +## 18.1 `GCP_TASKS_PROJECT_ID` + +Optional. + +Defaults to: + +```text +GCP_PROJECT_ID +``` + +## 18.2 `GCP_TASKS_LOCATION` + +Required. + +Example: + +```text +GCP_TASKS_LOCATION=us-central1 +``` + +## 18.3 `GCP_TASKS_QUEUE` + +Required. + +Example: + +```text +GCP_TASKS_QUEUE=care-default +``` + +Multiple queues MAY later use task-class-specific variables. + +## 18.4 `GCP_WORKER_URL` + +Required. + +Example: + +```text +GCP_WORKER_URL=https://care-prod-worker-...run.app/internal/tasks/execute/ +``` + +The URL SHALL target the private worker service. + +## 18.5 `GCP_TASKS_SERVICE_ACCOUNT` + +Required. + +This is the service account identity attached to OIDC task requests. + +Example: + +```text +care-tasks-invoker@care-production.iam.gserviceaccount.com +``` + +## 18.6 `GCP_TASKS_OIDC_AUDIENCE` + +Recommended. + +Often equal to the worker service origin. + +It SHALL match worker IAM expectations. + +## 18.7 `GCP_TASKS_DEFAULT_DEADLINE_SECONDS` + +Optional. + +The value SHALL remain within Cloud Tasks and Cloud Run supported limits. + +Different queues MAY use different infrastructure-level deadlines. + +## 18.8 `GCP_TASKS_DEFAULT_QUEUE` + +Optional alias for `GCP_TASKS_QUEUE`. + +The project SHOULD avoid maintaining redundant names indefinitely. + +## 18.9 Retry configuration + +Retry policy SHOULD be managed in Terraform. + +Examples: + +```text +max attempts +max retry duration +minimum backoff +maximum backoff +maximum doublings +``` + +Task code SHALL not silently override infrastructure policy without +documentation. + +--- + +# 19. Cloud Tasks Worker Configuration + +## 19.1 `CARE_TASK_HANDLER_ENDPOINT_ENABLED` + +Optional. + +Recommended values: + +```text +API role: false +task_worker role: true +``` + +This variable MAY prevent internal task routes from being exposed by the public +API service. + +## 19.2 `CARE_TASK_ALLOWED_QUEUE_NAMES` + +Optional defense-in-depth configuration. + +The worker MAY validate expected Cloud Tasks queue headers. + +This SHALL not replace IAM. + +## 19.3 `CARE_TASK_LOG_PAYLOAD` + +Production-required value: + +```text +false +``` + +Full task payload logging is prohibited. + +## 19.4 `CARE_TASK_HANDLER_TIMEOUT_SECONDS` + +Optional application-level timeout. + +It SHALL remain lower than the infrastructure request deadline when enforced. + +## 19.5 `CARE_TASK_RETRYABLE_EXCEPTIONS` + +SHOULD be defined in code, not as arbitrary import paths from environment +variables. + +Configuration MAY control categories, but SHALL not enable arbitrary code +loading. + +--- + +# 20. Celery Configuration + +Used when: + +```text +CARE_TASK_BACKEND=celery +``` + +## 20.1 `CELERY_BROKER_URL` + +Required. + +Common local value: + +```text +redis://redis:6379/0 +``` + +## 20.2 `CELERY_RESULT_BACKEND` + +Optional depending on CARE call-site requirements. + +Existing local compatibility MAY use the broker URL. + +## 20.3 `CELERY_TASK_ALWAYS_EAGER` + +Test-only option. + +SHALL not be enabled in production unintentionally. + +## 20.4 `CELERY_BEAT_ENABLED` + +Recommended explicit variable. + +Local traditional value: + +```text +true +``` + +GCP value: + +```text +false +``` + +The GCP profile SHALL not run Celery Beat. + +## 20.5 `CELERY_WORKER_CONCURRENCY` + +Optional. + +Must respect database and Redis connection budgets. + +--- + +# 21. PostgreSQL Task Queue Configuration + +This section applies only if the optional queue backend is approved. + +## 21.1 `CARE_POSTGRES_QUEUE_SCHEMA` + +Optional. + +Example: + +```text +CARE_POSTGRES_QUEUE_SCHEMA=care_tasks +``` + +## 21.2 `CARE_POSTGRES_QUEUE_NAMES` + +Optional list. + +Example: + +```text +CARE_POSTGRES_QUEUE_NAMES=["default","reports","email"] +``` + +Do not create multiple queues without a workload reason. + +## 21.3 `CARE_POSTGRES_WORKER_CONCURRENCY` + +Optional. + +Conservative default: + +```text +1 +``` + +## 21.4 `CARE_POSTGRES_WORKER_POLL_INTERVAL` + +Optional. + +Only relevant if the chosen queue implementation polls. + +## 21.5 `CARE_POSTGRES_WORKER_HEARTBEAT_SECONDS` + +Optional. + +Used for worker-health diagnostics. + +## 21.6 `CARE_POSTGRES_JOB_RETENTION_DAYS` + +Optional. + +Defines retention for completed or failed task records. + +## 21.7 `CARE_POSTGRES_QUEUE_ENABLED` + +May be redundant with `CARE_TASK_BACKEND=postgres`. + +The implementation SHOULD prefer one authoritative switch. + +--- + +# 22. Cache Backend Selection + +## 22.1 `CARE_CACHE_BACKEND` + +Required. + +Supported values: + +```text +postgres +locmem +redis +dummy +``` + +Recommended low-cost GCP value: + +```text +CARE_CACHE_BACKEND=postgres +``` + +Local compatibility value: + +```text +CARE_CACHE_BACKEND=redis +``` + +Test value: + +```text +CARE_CACHE_BACKEND=dummy +``` + +or: + +```text +CARE_CACHE_BACKEND=locmem +``` + +depending on test behavior. + +--- + +# 23. PostgreSQL Cache Configuration + +Required when: + +```text +CARE_CACHE_BACKEND=postgres +``` + +## 23.1 `CARE_CACHE_TABLE` + +Recommended. + +Default: + +```text +care_cache +``` + +## 23.2 `CARE_CACHE_TIMEOUT` + +Optional. + +Defines the default Django cache timeout. + +Example: + +```text +CARE_CACHE_TIMEOUT=300 +``` + +## 23.3 `CARE_CACHE_MAX_ENTRIES` + +Optional. + +Example: + +```text +CARE_CACHE_MAX_ENTRIES=10000 +``` + +The final value SHALL be based on measured usage. + +## 23.4 `CARE_CACHE_CULL_FREQUENCY` + +Optional. + +Example: + +```text +CARE_CACHE_CULL_FREQUENCY=3 +``` + +## 23.5 `CARE_CACHE_KEY_PREFIX` + +Recommended. + +Example: + +```text +care:prod +``` + +This helps separate environments or logical uses. + +## 23.6 `CARE_CACHE_VERSION` + +Optional Django cache version value. + +## 23.7 Table creation + +The application SHALL fail clearly or report unhealthy configuration when the +selected database cache table does not exist. + +It SHALL not create the table during ordinary request startup. + +--- + +# 24. LocMem Cache Configuration + +## 24.1 `CARE_CACHE_LOCATION` + +Optional. + +Example: + +```text +care-local-cache +``` + +## 24.2 `CARE_CACHE_MAX_ENTRIES` + +Optional. + +LocMem remains process-local and ephemeral. + +The configuration SHALL not imply cross-instance consistency. + +--- + +# 25. Redis Cache Configuration + +Required when: + +```text +CARE_CACHE_BACKEND=redis +``` + +## 25.1 `REDIS_CACHE_URL` + +Required secret. + +Example: + +```text +rediss://default:@:6379/0 +``` + +## 25.2 `REDIS_CACHE_PREFIX` + +Recommended. + +Example: + +```text +care:prod:cache +``` + +## 25.3 `REDIS_CACHE_TIMEOUT` + +Optional. + +## 25.4 `REDIS_CACHE_SOCKET_TIMEOUT` + +Recommended. + +## 25.5 `REDIS_CACHE_CONNECT_TIMEOUT` + +Recommended. + +## 25.6 `REDIS_CACHE_IGNORE_EXCEPTIONS` + +Default SHOULD depend on cache purpose. + +For performance-only cache: + +```text +true +``` + +may be acceptable. + +For correctness-sensitive state: + +```text +false +``` + +or a dedicated non-cache backend is preferred. + +--- + +# 26. Rate-Limit Backend Selection + +## 26.1 `CARE_RATE_LIMIT_BACKEND` + +Required. + +Supported values: + +```text +postgres +redis +``` + +Recommended Redis-free GCP value: + +```text +CARE_RATE_LIMIT_BACKEND=postgres +``` + +LocMem SHALL not be a supported globally consistent production backend. + +## 26.2 `CARE_RATE_LIMIT_DEFAULT` + +Existing CARE-compatible default MAY be: + +```text +5/10m +``` + +The exact syntax SHALL remain compatible with the selected library. + +## 26.3 `DISABLE_RATELIMIT` + +Production-required value: + +```text +false +``` + +Disabling rate limiting in production SHALL require an explicit exceptional +decision. + +## 26.4 `CARE_RATE_LIMIT_FAILURE_POLICY` + +Recommended. + +Supported conceptual values: + +```text +fail_closed +fail_open +controlled_error +``` + +Security-sensitive endpoints SHOULD not silently fail open. + +--- + +# 27. PostgreSQL Rate-Limit Configuration + +## 27.1 `CARE_RATE_LIMIT_TABLE` + +Optional. + +Required only if explicit database models or tables are introduced. + +## 27.2 `CARE_RATE_LIMIT_RETENTION_SECONDS` + +Optional. + +Defines cleanup retention for expired counters. + +## 27.3 `CARE_RATE_LIMIT_CLEANUP_BATCH_SIZE` + +Optional. + +Used by maintenance jobs where applicable. + +--- + +# 28. Redis Rate-Limit Configuration + +Required when: + +```text +CARE_RATE_LIMIT_BACKEND=redis +``` + +## 28.1 `REDIS_RATE_LIMIT_URL` + +Required secret. + +It MAY equal `REDIS_CACHE_URL`, but SHALL be configurable independently. + +## 28.2 `REDIS_RATE_LIMIT_PREFIX` + +Recommended. + +Example: + +```text +care:prod:ratelimit +``` + +## 28.3 `REDIS_RATE_LIMIT_SOCKET_TIMEOUT` + +Recommended. + +## 28.4 Failure behavior + +The configured outage policy SHALL be tested. + +--- + +# 29. Transient-State Backend Selection + +## 29.1 `CARE_TRANSIENT_STATE_BACKEND` + +Required. + +Supported values: + +```text +postgres +redis +``` + +Recommended low-cost GCP value: + +```text +CARE_TRANSIENT_STATE_BACKEND=postgres +``` + +## 29.2 Durable versus disposable state + +The variable SHALL select shared short-lived state only. + +Correctness-critical or auditable state SHOULD use explicit PostgreSQL models +regardless of cache backend. + +--- + +# 30. PostgreSQL Transient-State Configuration + +## 30.1 `CARE_TRANSIENT_STATE_TABLE` + +Optional. + +Used only if a dedicated generic state table is implemented. + +A generic table SHOULD not replace domain-specific models without need. + +## 30.2 `CARE_TRANSIENT_STATE_DEFAULT_TTL` + +Optional. + +## 30.3 `CARE_TRANSIENT_STATE_CLEANUP_BATCH_SIZE` + +Optional. + +--- + +# 31. Redis Transient-State Configuration + +## 31.1 `REDIS_TRANSIENT_STATE_URL` + +Required when Redis transient state is selected. + +## 31.2 `REDIS_TRANSIENT_STATE_PREFIX` + +Recommended. + +Example: + +```text +care:prod:state +``` + +## 31.3 `REDIS_TRANSIENT_STATE_DEFAULT_TTL` + +Optional. + +--- + +# 32. Report Progress Configuration + +## 32.1 `CARE_REPORT_PROGRESS_BACKEND` + +Recommended explicit variable. + +Supported values: + +```text +database_model +cache +``` + +When: + +```text +cache +``` + +is selected, the configured shared cache backend is used. + +When: + +```text +database_model +``` + +is selected, durable task or report-progress records are used. + +## 32.2 Recommended initial value + +If users need reliable cross-instance visibility and failure history: + +```text +CARE_REPORT_PROGRESS_BACKEND=database_model +``` + +If disposable progress is sufficient: + +```text +CARE_REPORT_PROGRESS_BACKEND=cache +``` + +## 32.3 `CARE_REPORT_PROGRESS_TIMEOUT` + +Required only for cache-backed progress. + +Example: + +```text +CARE_REPORT_PROGRESS_TIMEOUT=600 +``` + +The current two-minute behavior MAY be too short for real report generation and +SHALL be reviewed. + +--- + +# 33. Optional Redis Provider Configuration + +## 33.1 Provider-neutral configuration + +The application SHOULD not require a provider name. + +Standard Redis URLs should be sufficient. + +## 33.2 `REDIS_SSL_CERT_REQS` + +Optional. + +Production SHOULD verify certificates. + +Disabling verification SHALL require explicit justification. + +## 33.3 `REDIS_MAX_CONNECTIONS` + +Recommended. + +The value SHALL respect provider plan limits and Cloud Run scaling. + +## 33.4 `REDIS_HEALTH_CHECK_INTERVAL` + +Optional. + +## 33.5 `REDIS_RETRY_ON_TIMEOUT` + +Optional. + +Behavior SHALL be selected per responsibility. + +## 33.6 Upstash + +An Upstash deployment MAY configure: + +```text +REDIS_CACHE_URL=rediss://... +REDIS_RATE_LIMIT_URL=rediss://... +REDIS_TRANSIENT_STATE_URL=rediss://... +``` + +No `USE_UPSTASH` variable is required. + +--- + +# 34. Email Configuration + +## 34.1 `EMAIL_BACKEND` + +Default production value MAY remain: + +```text +django.core.mail.backends.smtp.EmailBackend +``` + +## 34.2 `EMAIL_HOST` + +Required when SMTP is enabled. + +## 34.3 `EMAIL_PORT` + +Required. + +## 34.4 `EMAIL_HOST_USER` + +Secret or protected value. + +## 34.5 `EMAIL_HOST_PASSWORD` + +Required secret when provider authentication requires it. + +## 34.6 `EMAIL_USE_TLS` + +Recommended according to provider requirements. + +## 34.7 `EMAIL_USE_SSL` + +Mutually constrained with TLS according to Django behavior. + +## 34.8 `DEFAULT_FROM_EMAIL` + +Required. + +Example: + +```text +CARE +``` + +## 34.9 `SERVER_EMAIL` + +Recommended for framework-generated error notifications where used. + +## 34.10 `CARE_EMAIL_TASK_QUEUE` + +Optional. + +May select a dedicated queue name when task isolation is implemented. + +--- + +# 35. Sentry Configuration + +## 35.1 `SENTRY_DSN` + +Optional secret. + +If absent, Sentry SHALL remain disabled. + +## 35.2 `SENTRY_ENVIRONMENT` + +Recommended. + +Defaults to: + +```text +CARE_ENVIRONMENT +``` + +## 35.3 `SENTRY_TRACES_SAMPLE_RATE` + +Optional. + +Production value SHALL be chosen with privacy and cost considerations. + +## 35.4 `SENTRY_PROFILES_SAMPLE_RATE` + +Optional. + +## 35.5 `SENTRY_EVENT_LEVEL` + +Optional. + +## 35.6 Integration selection + +The GCP settings SHALL enable integrations according to active backends: + +```text +Django integration -> normally enabled +Celery integration -> only when Celery is used +Redis integration -> only when Redis is used +``` + +--- + +# 36. Logging Configuration + +## 36.1 `CARE_LOG_FORMAT` + +Supported values SHOULD include: + +```text +json +text +``` + +Recommended GCP value: + +```text +json +``` + +## 36.2 `CARE_LOG_LEVEL` + +Recommended default: + +```text +INFO +``` + +Production `DEBUG` logging SHALL not be enabled broadly without review. + +## 36.3 `CARE_LOG_REQUEST_BODIES` + +Production-required value: + +```text +false +``` + +## 36.4 `CARE_LOG_TASK_PAYLOADS` + +Production-required value: + +```text +false +``` + +## 36.5 `CARE_LOG_FILE_CONTENTS` + +Production-required value: + +```text +false +``` + +## 36.6 `CARE_LOG_SQL` + +Production default: + +```text +false +``` + +Temporary SQL logging MAY be enabled in controlled non-production +environments. + +## 36.7 `CARE_REQUEST_ID_HEADER` + +Optional. + +Example: + +```text +X-Request-ID +``` + +--- + +# 37. Health-Check Configuration + +## 37.1 `CARE_HEALTH_DATABASE_ENABLED` + +Recommended: + +```text +true +``` + +## 37.2 `CARE_HEALTH_CACHE_ENABLED` + +Recommended when the selected cache is required for normal operation. + +## 37.3 `CARE_HEALTH_REDIS_ENABLED` + +SHALL default according to active Redis responsibilities. + +It SHALL not be required in a Redis-free profile. + +## 37.4 `CARE_HEALTH_CELERY_ENABLED` + +GCP Cloud Tasks profile: + +```text +false +``` + +Local Celery profile: + +```text +true +``` + +## 37.5 `CARE_HEALTH_POSTGRES_QUEUE_ENABLED` + +Only when the PostgreSQL queue backend is selected. + +## 37.6 `CARE_HEALTH_STORAGE_ENABLED` + +Optional. + +A storage diagnostic check MAY be useful. + +It SHALL avoid writing test objects on every public health request. + +## 37.7 `CARE_HEALTH_PUBLIC_DETAILS` + +Production-required value: + +```text +false +``` + +Detailed dependency diagnostics SHOULD require operator authorization. + +--- + +# 38. Cloud Run Job Configuration + +## 38.1 `CARE_JOB_NAME` + +Recommended log metadata. + +Examples: + +```text +migrate +sync-permissions +cleanup-incomplete-uploads +``` + +## 38.2 `CARE_JOB_COMMAND` + +Prefer command configuration through the Cloud Run Job container command +rather than arbitrary runtime shell execution. + +## 38.3 `CARE_JOB_TIMEOUT_SECONDS` + +Infrastructure-level setting managed by Terraform. + +## 38.4 `CARE_JOB_MAX_RETRIES` + +Infrastructure-level setting. + +## 38.5 `CARE_JOB_DRY_RUN` + +Optional for commands supporting non-destructive previews. + +Destructive jobs SHOULD support a dry-run mode where practical. + +--- + +# 39. Cloud Scheduler Configuration + +Scheduler configuration SHOULD primarily live in Terraform. + +Per schedule, define: + +```text +name +cron expression +timezone +target +authentication +retry policy +enabled state +``` + +## 39.1 `CARE_SCHEDULER_TIMEZONE` + +Optional shared default. + +The timezone SHALL be explicit. + +It SHALL not inherit an unrelated Celery timezone accidentally. + +## 39.2 Cleanup cadence + +Variables MAY control schedule creation, but Terraform remains authoritative. + +Examples: + +```text +CARE_EXPIRED_TOKEN_CLEANUP_CRON +CARE_INCOMPLETE_UPLOAD_CLEANUP_CRON +``` + +The application SHALL not dynamically register GCP production schedules at +startup. + +--- + +# 40. Authentication and JWT Configuration + +CARE's existing authentication configuration SHALL remain authoritative. + +Relevant secrets and variables may include: + +```text +JWKS_BASE64 +JWT-related keys +token lifetimes +issuer and audience settings +``` + +These SHALL be preserved during GCP adaptation. + +Private signing material SHALL be stored in Secret Manager. + +Authentication behavior SHALL not change merely because the application moves +to Cloud Run. + +--- + +# 41. External Service Configuration + +Existing CARE integrations MAY require variables for: + +```text +Snowstorm or terminology services +SMS providers +SMTP +Sentry +other plugins +``` + +Each integration SHALL define: + +- required variables; +- whether values are secret; +- timeout; +- failure behavior; +- health-check behavior; +- enabled state. + +Optional integrations SHALL not prevent API startup when disabled. + +--- + +# 42. Plugin Configuration + +Plugins may add environment variables and dependencies. + +Required production plugins SHALL be inventoried. + +A plugin SHALL not be enabled without confirming compatibility with: + +- GCP settings; +- Django Storage API; +- Cloud Tasks or selected task backend; +- Redis-free operation; +- Cloud Run startup; +- empty-database initialization. + +Plugin configuration SHALL not be mixed into the core GCP contract without a +documented reason. + +--- + +# 43. Role-Specific Variable Matrix + +| Variable group | API | HTTP worker | Jobs | PostgreSQL queue worker | Celery worker | +|---|---:|---:|---:|---:|---:| +| Django core | Required | Required | Required | Required | Required | +| Database | Required | Required | Required | Required | Required | +| Storage | Required | As handlers require | As commands require | As tasks require | As tasks require | +| Cloud Tasks enqueue | Usually required | Optional | Optional | No | No | +| Cloud Tasks worker URL | Required for enqueue | No | Optional | No | No | +| Task handler endpoint | No | Required | No | No | No | +| PostgreSQL queue config | No unless producer | No | Optional | Required | No | +| Celery broker | No | No | No | No | Required | +| Redis cache | Only if selected | Only if selected | Only if selected | Only if selected | Often | +| Email | As required | As required | Rarely | As tasks require | As tasks require | + +Variables SHALL be injected only where needed where practical. + +--- + +# 44. Default GCP Profile + +Recommended initial configuration: + +```text +DJANGO_SETTINGS_MODULE=config.settings.deployment +CARE_ENVIRONMENT=prod +DJANGO_DEBUG=false + +CARE_PROCESS_ROLE=api + +CARE_STORAGE_BACKEND=gcs +CARE_TASK_BACKEND=cloud_tasks +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +CARE_REPORT_PROGRESS_BACKEND=database_model + +GCP_PROJECT_ID= +GCP_REGION= + +CARE_PATIENT_STORAGE_BUCKET= +CARE_FACILITY_STORAGE_BUCKET= +CARE_REPORT_STORAGE_BUCKET= + +GCP_TASKS_LOCATION= +GCP_TASKS_QUEUE= +GCP_WORKER_URL= +GCP_TASKS_SERVICE_ACCOUNT= + +CARE_LOG_FORMAT=json +CARE_LOG_LEVEL=INFO +CARE_LOG_REQUEST_BODIES=false +CARE_LOG_TASK_PAYLOADS=false +``` + +Secrets are injected separately. + +This profile SHALL start without any Redis variable. + +--- + +# 45. Local Upstream-Compatible Profile + +Conceptual local configuration: + +```text +DJANGO_SETTINGS_MODULE=config.settings.local +CARE_ENVIRONMENT=dev + +CARE_STORAGE_BACKEND=s3 +CARE_TASK_BACKEND=celery +CARE_CACHE_BACKEND=redis +CARE_RATE_LIMIT_BACKEND=redis +CARE_TRANSIENT_STATE_BACKEND=redis + +BUCKET_ENDPOINT=http://minio:9000 +BUCKET_KEY=minioadmin +BUCKET_SECRET=minioadmin +FILE_UPLOAD_BUCKET=patient-bucket +FACILITY_S3_BUCKET=facility-bucket + +CELERY_BROKER_URL=redis://redis:6379/0 +CELERY_RESULT_BACKEND=redis://redis:6379/0 + +REDIS_CACHE_URL=redis://redis:6379/1 +REDIS_RATE_LIMIT_URL=redis://redis:6379/2 +REDIS_TRANSIENT_STATE_URL=redis://redis:6379/3 +``` + +The storage variables are the pre-existing ones documented in §13; `S3_*` names +are not accepted. The shared `BUCKET_*` values serve every alias unless an +alias-specific `FILE_UPLOAD_*` or `FACILITY_S3_*` value overrides them. + +The implementation MAY preserve existing local `REDIS_URL` compatibility while +introducing more specific variables gradually. + +Development defaults SHALL not flow into production settings. + +--- + +# 46. Optional Upstash Profile + +Conceptual example: + +```text +CARE_TASK_BACKEND=cloud_tasks +CARE_CACHE_BACKEND=redis +CARE_RATE_LIMIT_BACKEND=redis +CARE_TRANSIENT_STATE_BACKEND=redis + +REDIS_CACHE_URL=rediss://... +REDIS_RATE_LIMIT_URL=rediss://... +REDIS_TRANSIENT_STATE_URL=rediss://... +``` + +The same URL MAY be reused initially. + +The application SHALL not assume separate physical databases are supported or +necessary without checking the provider. + +Namespaces or key prefixes SHALL isolate responsibilities. + +--- + +# 47. Consolidated PostgreSQL Profile + +Only if the PostgreSQL queue backend is approved: + +```text +CARE_TASK_BACKEND=postgres +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +``` + +This profile requires: + +- queue schema; +- queue worker; +- worker health monitoring; +- task-record retention; +- additional Cloud SQL capacity planning. + +It SHALL not be labeled scale-to-zero when immediate task execution requires an +active worker. + +--- + +# 48. Test Profile + +Conceptual fast test configuration: + +```text +CARE_TASK_BACKEND=fake +CARE_CACHE_BACKEND=dummy +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +``` + +`CARE_STORAGE_BACKEND` is deliberately absent. Its only accepted values are `s3` +and `gcs` (§11.2); `filesystem` is not one of them and would raise +`ImproperlyConfigured` at startup. Tests that must avoid a real bucket +substitute `django.core.files.storage.InMemoryStorage` through +`override_settings` on `STORAGES`, which is a test-local override rather than a +configuration value. + +A `fake` task backend MAY exist only in test settings. + +Production settings SHALL reject it. + +Provider integration tests SHALL explicitly override the fake backends. + +--- + +# 49. Deprecated Compatibility Variables + +During implementation, CARE may temporarily continue accepting existing +variables such as: + +```text +REDIS_URL +BUCKET_PROVIDER +BUCKET_REGION +BUCKET_KEY +BUCKET_SECRET +BUCKET_ENDPOINT +BUCKET_EXTERNAL_ENDPOINT +FILE_UPLOAD_BUCKET +FILE_UPLOAD_BUCKET_ENDPOINT +FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT +FACILITY_S3_BUCKET +FACILITY_S3_BUCKET_ENDPOINT +FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT +``` + +Compatibility behavior SHALL: + +- emit deprecation warnings where safe; +- map old values to new settings only when unambiguous; +- avoid mixing old and new values silently; +- define precedence clearly; +- document eventual removal. + +Because the production deployment is greenfield, new GCP environments SHOULD +use only the new variables. + +--- + +# 50. Configuration Precedence + +Recommended precedence: + +1. explicit new configuration variable; +2. supported compatibility variable; +3. safe non-secret default; +4. configuration error. + +If both new and old variables are set with conflicting values: + +- the new value MAY take precedence; +- startup SHOULD emit a warning; +- production MAY reject the conflict to avoid ambiguity. + +Secret values SHALL never be printed in warnings. + +--- + +# 51. Safe Defaults + +Safe defaults MAY exist for: + +```text +log level +cache timeout +task delay +process role in local development +non-secret feature toggles +``` + +Defaults SHALL not exist for production: + +```text +Django secret key +database password +SMTP password +Redis password +private signing keys +service-account credentials +production bucket names +production trusted origins +``` + +--- + +# 52. Prohibited Production Defaults + +The GCP settings SHALL reject or warn critically about: + +```text +DJANGO_DEBUG=true +DJANGO_ALLOWED_HOSTS=* +DISABLE_RATELIMIT=true +public storage configuration +default MinIO credentials +localhost database URL +localhost Redis URL +local MinIO endpoint +service-account JSON bundled in image +CARE_LOG_REQUEST_BODIES=true +CARE_LOG_TASK_PAYLOADS=true +``` + +The exact enforcement MAY differ between `dev`, `staging` and `prod`. + +--- + +# 53. Startup Validation + +At startup, CARE SHOULD validate: + +- selected backend names; +- required variables for each backend; +- mutually incompatible settings; +- role-specific requirements; +- production security values; +- storage aliases; +- cache table configuration; +- worker URL format; +- queue location and name; +- Redis URL scheme when Redis is selected. + +Startup validation SHALL avoid making destructive calls. + +External connectivity checks belong in readiness or diagnostics, not settings +parsing. + +--- + +# 54. Configuration Diagnostics + +An authorized management command SHOULD display effective non-secret +configuration. + +Conceptual command: + +```bash +python manage.py check_gcp_configuration +``` + +It MAY report: + +```text +environment +process role +storage backend and aliases +task backend +cache backend +rate-limit backend +transient-state backend +report-progress backend +required service availability +health-check selection +``` + +It SHALL redact: + +```text +passwords +secret keys +tokens +complete URLs containing credentials +private key material +``` + +--- + +# 55. Configuration Test Matrix + +At minimum, automated tests SHALL validate: + +| Profile | Tasks | Cache | Rate limits | State | Storage | +|---|---|---|---|---|---| +| GCP default | Cloud Tasks | PostgreSQL | PostgreSQL | PostgreSQL | GCS | +| GCP Redis | Cloud Tasks | Redis | Redis | Redis | GCS | +| Local | Celery | Redis | Redis | Redis | MinIO/S3 | +| GCP LocMem | Cloud Tasks | LocMem | PostgreSQL | PostgreSQL | GCS | +| Consolidated PostgreSQL | PostgreSQL queue | PostgreSQL | PostgreSQL | PostgreSQL | configured storage | +| Test | fake/eager | Dummy | test backend | test backend | filesystem | + +The consolidated profile applies only if implemented. + +--- + +# 56. Configuration Change Procedure + +Before changing production configuration: + +1. identify affected services; +2. determine whether a new revision is required; +3. determine whether the value is secret; +4. test in staging; +5. review IAM and dependency implications; +6. deploy the new configuration; +7. run smoke tests; +8. monitor logs and metrics; +9. record the change. + +Changing a backend value may require additional resources. + +Example: + +```text +CARE_CACHE_BACKEND=postgres -> redis +``` + +requires a valid Redis service and secret. + +--- + +# 57. Backend Change Semantics + +## Cache backend + +Cache values are disposable. + +Changing cache backend does not require state migration. + +## Rate-limit backend + +Changing backend resets or separates counters unless a deliberate state +transfer is implemented. + +The operational effect SHALL be understood. + +## Transient-state backend + +Existing temporary state may become unavailable after a switch. + +Correctness-critical state SHALL not rely on an unplanned backend switch. + +## Task backend + +Queued tasks do not automatically move between backends. + +Backend changes SHALL occur only when the previous queue is empty or its +remaining work is intentionally handled. + +For the initial greenfield launch, no legacy production queue exists. + +## Storage backend + +Storage objects do not automatically move between providers. + +The greenfield GCP launch starts with empty GCS buckets. + +After real use begins, changing storage requires a separate migration plan. + +--- + +# 58. Configuration Ownership + +Each configuration group SHOULD have an owner. + +Suggested ownership: + +```text +Django security -> application maintainers +Cloud Run -> platform maintainers +Cloud SQL -> database/platform maintainers +storage -> application and platform maintainers +tasks -> application and platform maintainers +Redis -> platform maintainers +email -> application operations +secrets -> security/platform maintainers +``` + +Ownership MAY be held by the same person in a small deployment, but +responsibilities SHALL remain explicit. + +--- + +# 59. Configuration Documentation Requirements + +Each new variable SHALL document: + +- name; +- purpose; +- whether required; +- whether secret; +- supported values; +- default; +- applicable roles; +- applicable environments; +- validation behavior; +- operational impact. + +Undocumented production variables SHALL not be introduced casually. + +--- + +# 60. Example API Service Configuration + +Non-secret conceptual values: + +```text +DJANGO_SETTINGS_MODULE=config.settings.deployment +CARE_ENVIRONMENT=prod +CARE_PROCESS_ROLE=api +DJANGO_DEBUG=false + +CARE_STORAGE_BACKEND=gcs +CARE_TASK_BACKEND=cloud_tasks +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +CARE_REPORT_PROGRESS_BACKEND=database_model + +GCP_PROJECT_ID=care-production +GCP_REGION=us-central1 +GCP_TASKS_LOCATION=us-central1 +GCP_TASKS_QUEUE=care-default +GCP_WORKER_URL=https://care-prod-worker-...run.app/internal/tasks/execute/ +GCP_TASKS_SERVICE_ACCOUNT=care-tasks-invoker@care-production.iam.gserviceaccount.com + +CARE_PATIENT_STORAGE_BUCKET=care-prod-patient-files +CARE_FACILITY_STORAGE_BUCKET=care-prod-facility-files +CARE_REPORT_STORAGE_BUCKET=care-prod-reports + +CARE_CACHE_TABLE=care_cache +CARE_LOG_FORMAT=json +CARE_LOG_LEVEL=INFO +``` + +Secrets: + +```text +DJANGO_SECRET_KEY +DATABASE_URL +EMAIL_HOST_PASSWORD +JWKS_BASE64 or equivalent private material +SENTRY_DSN, when enabled +``` + +--- + +# 61. Example Worker Service Configuration + +```text +DJANGO_SETTINGS_MODULE=config.settings.deployment +CARE_ENVIRONMENT=prod +CARE_PROCESS_ROLE=task_worker + +CARE_STORAGE_BACKEND=gcs +CARE_TASK_BACKEND=cloud_tasks +CARE_CACHE_BACKEND=postgres +CARE_RATE_LIMIT_BACKEND=postgres +CARE_TRANSIENT_STATE_BACKEND=postgres +CARE_REPORT_PROGRESS_BACKEND=database_model + +CARE_TASK_HANDLER_ENDPOINT_ENABLED=true +CARE_TASK_LOG_PAYLOAD=false + +GCP_PROJECT_ID=care-production +GCP_REGION=us-central1 + +CARE_PATIENT_STORAGE_BUCKET=care-prod-patient-files +CARE_FACILITY_STORAGE_BUCKET=care-prod-facility-files +CARE_REPORT_STORAGE_BUCKET=care-prod-reports + +CARE_LOG_FORMAT=json +CARE_LOG_LEVEL=INFO +``` + +The worker does not necessarily need queue-enqueue configuration unless tasks +can create follow-up tasks. + +--- + +# 62. Example Migration Job Configuration + +```text +DJANGO_SETTINGS_MODULE=config.settings.deployment +CARE_ENVIRONMENT=prod +CARE_PROCESS_ROLE=job +CARE_JOB_NAME=migrate + +CARE_STORAGE_BACKEND=gcs +CARE_CACHE_BACKEND=postgres + +GCP_PROJECT_ID=care-production +GCP_REGION=us-central1 + +CARE_LOG_FORMAT=json +CARE_LOG_LEVEL=INFO +``` + +The migration job may not require task-dispatch configuration. + +The exact settings validation SHALL account for process role. + +--- + +# 63. Definition of Configuration Completion + +Configuration implementation is complete when: + +- GCP settings load with explicit validated values; +- the default GCP profile starts without Redis; +- local Celery, Redis and MinIO remain supported; +- GCS storage aliases resolve; +- MinIO aliases resolve locally; +- Cloud Tasks variables are required only when selected; +- PostgreSQL cache variables are validated; +- optional Redis responsibilities use independent URLs; +- role-specific services receive only required configuration; +- production rejects insecure defaults; +- diagnostics redact secrets; +- tests cover supported profile combinations; +- configuration documentation matches implementation. + +--- + +## 64. Next Document + +The next document is: + +```text +docs/xii/architecture/08-terraform-architecture.md +``` + +It will define: + +- Terraform repository structure; +- environment composition; +- modules; +- APIs; +- service accounts; +- IAM; +- Cloud SQL; +- Cloud Storage; +- Artifact Registry; +- Cloud Run services; +- Cloud Run Jobs; +- Cloud Tasks; +- Cloud Scheduler; +- Secret Manager; +- monitoring; +- state management; +- outputs; +- resource-protection rules. diff --git a/docs/xii/architecture/inventory/cache-and-redis.md b/docs/xii/architecture/inventory/cache-and-redis.md new file mode 100644 index 0000000000..2d106eddee --- /dev/null +++ b/docs/xii/architecture/inventory/cache-and-redis.md @@ -0,0 +1,521 @@ +--- +title: Cache and Redis Inventory +document: inventory/cache-and-redis +version: 0.1.0 +status: Draft +phase: 0 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +reviewed: 2026-08-05 +--- + +# Cache and Redis Inventory + +Every cache and Redis use in the repository, classified by role, with a +backend-suitability assessment grounded in the semantics each site actually +requires. No cache or Redis code was modified in this phase. + +Evidence labels: **verified** / **inferred** / **unknown**. + +--- + +## 1. Summary + +**verified** Redis is used for **four structurally different things**, only one of +which is an ordinary cache: + +1. Celery broker and result backend (`base.py:421`, `:423`) +2. Django cache backend (`base.py:85-95`) +3. **Distributed locking via `SETNX`** (`care/utils/lock.py`) +4. **Redis LIST data structures via a raw client** (`care/emr/models/valueset.py`) + +**verified** Three distinct hard couplings prevent swapping the cache backend: + +| # | Coupling | Location | Why it blocks | +| --- | --- | --- | --- | +| A | `cache.set(..., nx=True)` | `care/utils/lock.py:18`, `:44` | `nx` is not in Django's cache API | +| B | `cache.delete_pattern(...)` | `care/emr/resources/base.py:313`, `:315` | `delete_pattern` is a `django_redis` extension | +| C | `get_redis_connection("default")` | `care/emr/models/valueset.py:77` | raw Redis client, LIST commands | + +**verified** These are not stylistic. Each one calls an API that does not exist on +`django.core.cache.backends.db.DatabaseCache` or `LocMemCache`. + +--- + +## 2. Configuration + +**verified** `config/settings/base.py`: + +```python +REDIS_URL = env("REDIS_URL", default="redis://localhost:6379") # line 80 + +CACHES = { # lines 85-100 + "default": { + "BACKEND": "django_redis.cache.RedisCache", # line 87 + "LOCATION": REDIS_URL, # line 88 + "OPTIONS": { + "CLIENT_CLASS": "django_redis.client.DefaultClient", # line 90 + "IGNORE_EXCEPTIONS": True, # line 93 + }, + }, + "swagger_cache": { + "BACKEND": "django.core.cache.backends.locmem.LocMemCache", # line 97 + "LOCATION": "swagger-schema-cache", # line 98 + }, +} +``` + +**verified** `IGNORE_EXCEPTIONS: True` (`base.py:93`) makes every cache operation +swallow connection errors and return `None`. + +**verified interaction with locking:** `Lock.acquire` (`care/utils/lock.py:17-19`) +treats a falsy return from `cache.set` as "lock already held" and raises +`ObjectLocked`. With `IGNORE_EXCEPTIONS: True`, a Redis outage makes `cache.set` +return `None`, so **every lock acquisition fails closed** and every locked +endpoint returns HTTP 423. **inferred** This means Redis is not merely a +performance dependency for those endpoints — it is a hard availability +dependency. + +**verified** `config/settings/test.py:44-50` overrides the default cache but still +uses `django_redis.cache.RedisCache` against `REDIS_URL` (`test.py:45-46`). The +test suite therefore requires a live Redis. + +**verified** `config/settings/test.py:58` silences `django_ratelimit.E003` and +`W001`. + +**verified** `LOCK_TIMEOUT = env.int("LOCK_TIMEOUT", default=32)` at +`config/settings/base.py:78`, commented `# timeout for setnx lock`. + +--- + +## 3. The `nx` shim and what it hides + +**verified** `config/caches.py` in full: + +```python +from django.core.cache.backends import dummy, locmem +from django.core.cache.backends.base import DEFAULT_TIMEOUT + + +class DummyCache(dummy.DummyCache): + def set(self, key, value, timeout=DEFAULT_TIMEOUT, version=None, nx=None): + super().set(key, value, timeout, version) + # mimic the behavior of django_redis with setnx, for tests + return True # line 9 + + +class LocMemCache(locmem.LocMemCache): + def set(self, key, value, timeout=DEFAULT_TIMEOUT, version=None, nx=None): + super().set(key, value, timeout, version) + # mimic the behavior of django_redis with setnx, for tests + return True # line 16 +``` + +**verified** Both subclasses accept `nx` and **ignore it**, unconditionally +returning `True` (`caches.py:9`, `caches.py:16`). + +**verified consequence:** under either shimmed backend, `Lock.acquire` +(`lock.py:17-19`) never raises, because `cache.set` always returns truthy. +**Mutual exclusion is silently disabled.** The comment calls this "mimic the +behavior of django_redis with setnx", but it does not mimic `SETNX` — it mimics +only the success case. + +**inferred** This is the central risk in "make Redis optional". A naive swap to +`DatabaseCache` or `LocMemCache` does not degrade locking gracefully; it removes +locking while leaving the call sites looking correct. Any PostgreSQL-backed lock +must be a real conditional insert (`INSERT ... ON CONFLICT DO NOTHING`, or +`pg_try_advisory_lock`), not a cache `set`. + +**verified** `config/caches.py` is referenced **exactly once**, and only from a +test: `care/utils/tests/test_utils.py:18` sets +`"BACKEND": "config.caches.LocMemCache"` in an override. No settings module — +`base.py`, `local.py`, `test.py`, `deployment.py`, `production.py`, +`staging.py` — points at either class. + +**inferred** So the shim exists solely to let one test run without Redis, and the +locking semantics it fakes are never exercised in production. It should be read as +evidence that a LocMem fallback was attempted and left incomplete, not as an +existing non-Redis path. + +--- + +## 4. Call-site classification + +Roles use the permitted vocabulary: `celery_broker`, `celery_result_backend`, +`performance_cache`, `shared_cache`, `report_progress`, `rate_limit`, +`distributed_lock`, `transient_state`, `session`, `health_check`, `direct_redis`, +`unknown`. + +Backend options: `PostgreSQL database cache`, `explicit PostgreSQL model`, +`LocMem`, `Redis-compatible backend`, `not applicable`. + +### 4.1 Broker and result backend + +| Site | Line | Role | Viable backend | +| --- | --- | --- | --- | +| `config/settings/base.py` | 421 | `celery_broker` | not applicable — replaced by Cloud Tasks, not re-hosted | +| `config/settings/base.py` | 423 | `celery_result_backend` | not applicable — **no consumer exists**, see `task-call-sites.md` §2 | + +**verified** No `AsyncResult` anywhere in the repository. The result backend is +written and never read. + +### 4.2 Distributed locking + +| Site | Line | Symbol | Role | +| --- | --- | --- | --- | +| `care/utils/lock.py` | 18 | `Lock.acquire` | `distributed_lock` | +| `care/utils/lock.py` | 22 | `Lock.release` | `distributed_lock` | +| `care/utils/lock.py` | 44 | `MultipleItemsLock.acquire` | `distributed_lock` | +| `care/utils/lock.py` | 51 | `MultipleItemsLock.release` | `distributed_lock` | + +**verified** `MultipleItemsLock.acquire` (`lock.py:42-47`) acquires keys in list +order and calls `self.release()` on the first failure (`:45`) before raising. + +**verified** Locks carry a TTL — `settings.LOCK_TIMEOUT`, default 32 s +(`base.py:78`), applied at `lock.py:18` and `:44`. + +**Viable backends:** + +- `Redis-compatible backend` — **works today**; this is the status quo. +- `explicit PostgreSQL model` — **viable**, and the only correct PostgreSQL + option. Requires a unique constraint plus `INSERT ... ON CONFLICT DO NOTHING` + to get real atomicity, and an explicit expiry column plus a sweeper, since + PostgreSQL has no native TTL. +- `PostgreSQL database cache` — **not viable as-is.** Django's `DatabaseCache` + exposes `add()`, which is atomic-ish, but not the `nx=` kwarg these call sites + pass. `cache.set(..., nx=True)` would raise `TypeError` on `DatabaseCache`. + Rewriting to `cache.add()` is plausible but changes the return contract and + needs verification against `DatabaseCache`'s `add()` implementation, which + performs a `SELECT` then an `INSERT` in a transaction rather than a single + atomic statement. +- `LocMem` — **not viable.** Per-process memory; provides no mutual exclusion + across Cloud Run instances. Under the shim it silently always succeeds. + +**verified** `care/security/management/commands/sync_permissions_roles.py:14` +documents this dependency in a docstring: *"multiple instances running the same +command is automatically blocked with redis"*. + +### 4.3 Raw Redis data structures + +| Site | Line | Symbol | Commands | Role | +| --- | --- | --- | --- | --- | +| `care/emr/models/valueset.py` | 77 | `RecentViewsManager.get_client` | `get_redis_connection("default")` | `direct_redis` | +| `care/emr/models/valueset.py` | 83 | `_remove_by_code` | `LRANGE` | `direct_redis` | +| `care/emr/models/valueset.py` | 89 | `_remove_by_code` | `LREM` | `direct_redis` | +| `care/emr/models/valueset.py` | 96 | `get_recent_views` | `LRANGE` | `direct_redis` | +| `care/emr/models/valueset.py` | 109 | `add_recent_view` | `LPUSH` | `direct_redis` | +| `care/emr/models/valueset.py` | 110 | `add_recent_view` | `LTRIM` | `direct_redis` | +| `care/emr/models/valueset.py` | 122 | `clear_recent_views` | `DEL` | `direct_redis` | + +**verified** This is a bounded most-recently-used list, capped by +`MAX_RECENT_VIEW` (`valueset.py:72`, default 20) enforced through +`LTRIM(key, 0, MAX-1)` at `:110`. + +**verified** `_client` is cached on the class (`valueset.py:71, 76-78`), so the +connection is created once per process. + +**Viable backends:** + +- `Redis-compatible backend` — works today. +- `explicit PostgreSQL model` — **viable.** A table keyed by user and valueset + with a timestamp reproduces the semantics: `LPUSH` + `LTRIM` becomes an insert + plus a delete of rows beyond rank 20; `LREM` by code becomes a delete by code. + This is a genuine schema addition, not a config change. +- `PostgreSQL database cache` — **not viable.** Django's cache API has no list + primitives. Emulating with read-modify-write on a JSON blob loses the atomicity + that `LPUSH`/`LTRIM` provide and would corrupt under concurrency. +- `LocMem` — **not viable.** Per-process; recent views would differ per Cloud Run + instance. + +**verified** `get_redis_connection` is imported from `django_redis` at +`valueset.py:5`. This import fails at module load if `django_redis` is absent, so +it is a hard package dependency, not just a runtime one. + +### 4.4 Pattern-based invalidation + +| Site | Line | Symbol | Role | +| --- | --- | --- | --- | +| `care/emr/resources/base.py` | 313 | model cache invalidation | `shared_cache` | +| `care/emr/resources/base.py` | 315 | model cache invalidation | `shared_cache` | + +**verified** Both lines call `cache.delete_pattern(...)`. + +**verified** `delete_pattern` is **not** part of `django.core.cache`. It is a +`django_redis` extension implemented with `SCAN` + `DEL`. Neither +`DatabaseCache` nor `LocMemCache` provides it; calling it raises +`AttributeError`. + +**verified** The paired writes are `cache.get` at `base.py:255` and `cache.set` at +`base.py:273`, keyed by `model_cache_key(model_string(db_model), model.__name__, pk)`. + +**Viable backends:** + +- `Redis-compatible backend` — works today. +- `PostgreSQL database cache` — **viable only after a code change.** `DatabaseCache` + stores keys in a real table, so a `LIKE`-based delete is expressible, but not + through the Django cache API. It needs either a custom backend subclass or + replacing pattern deletion with explicit key enumeration. +- `explicit PostgreSQL model` — not the natural fit; this is genuinely a cache. +- `LocMem` — **not viable** for correctness across instances: stale model data + would persist on every instance that did not serve the write. + +### 4.5 Ordinary performance caches + +**verified** These use only `get` / `set` / `delete` / `get_or_set` / +`delete_many` — all portable Django cache API. + +| File | Lines | Symbol | Role | +| --- | --- | --- | --- | +| `care/security/models/role.py` | 47, 57, 69, 87, 113, 115 | role permission caching | `performance_cache` | +| `care/emr/models/facility_config.py` | 50, 57, 62, 69, 76, 77 | monetary component / discount config | `performance_cache` | +| `care/emr/models/favorites.py` | 40, 44 | favorites | `performance_cache` | +| `care/emr/api/viewsets/favorites.py` | 40, 55, 102, 105 | favorites list | `shared_cache` | +| `care/emr/resources/favorites/filters.py` | 54, 68 | favorites filter | `performance_cache` | +| `care/emr/api/viewsets/valueset.py` | 124, 133, 155, 176, 193 | valueset favorites | `shared_cache` | +| `care/utils/models/base.py` | 51, 56, 74, 84 | flags cache (`get_or_set`) | `performance_cache` | +| `care/emr/fhir/resources/base.py` | 34, 37 | FHIR lookup, 10 s TTL (`:37`) | `performance_cache` | +| `care/users/api/viewsets/plug_config.py` | 17, 21, 26, 30, 34 | plug config response | `performance_cache` | +| `care/emr/resources/tag/cache_invalidation.py` | 31 | `delete_many` | `shared_cache` | + +**Viable backends for this group:** + +- `Redis-compatible backend` — works today. +- `PostgreSQL database cache` — **viable.** All operations are within the standard + Django cache API. Cost is a DB round trip per lookup and a `cache_table` that + needs `createcachetable` plus periodic culling. +- `LocMem` — **viable only for read-mostly, tolerant-of-staleness entries.** Not + viable for the `shared_cache` rows, where one instance's invalidation must be + seen by others. `care/emr/api/viewsets/favorites.py:102-105` and + `care/emr/resources/tag/cache_invalidation.py:31` delete keys after a write; + under LocMem, other instances keep serving stale data. + +**verified** `care/emr/resources/base.py:255, 273` belongs to this group in API +usage but is invalidated by §4.4's `delete_pattern`, so it inherits that blocker. + +### 4.6 Token invalidation + +| File | Line | Symbol | Role | +| --- | --- | --- | --- | +| `config/authentication.py` | 21 | `cache.get(ACCESS_TOKEN_INVALIDATION_PREFIX + ...)` | `transient_state` | +| `config/auth_views.py` | 99 | `cache.get(REFRESH_TOKEN_INVALIDATION_PREFIX + ...)` | `transient_state` | +| `config/auth_views.py` | 194, 199 | `cache.set(...)` | `transient_state` | + +**verified** This is a JWT denylist: tokens are checked against the cache on every +authenticated request (`authentication.py:21`). + +**Viable backends:** + +- `Redis-compatible backend` — works today. +- `PostgreSQL database cache` — **viable, with a caveat.** Correctness is fine, + but `authentication.py:21` runs on **every authenticated request**, so this + turns into an extra DB query per request. **inferred** measurable latency cost + on Cloud Run; worth benchmarking before committing. +- `explicit PostgreSQL model` — viable, and would allow indexing and a cleanup job. +- `LocMem` — **not viable.** A token revoked on one instance would remain valid on + every other instance. This is a security property, not a performance one. + +**verified** With `IGNORE_EXCEPTIONS: True` (`base.py:93`), a Redis outage makes +`cache.get` return `None` at `authentication.py:21`, which reads as +"not invalidated" — revoked tokens are accepted. Recorded in +`unresolved-items.md` §4. + +### 4.7 Report progress + +| File | Line | Symbol | Role | +| --- | --- | --- | --- | +| `care/emr/reports/report_utils.py` | 29 | `set_lock` → `cache.set(cache_key, progress, timeout)` | `report_progress` | +| `care/emr/reports/report_utils.py` | 34 | `get_progress` → `cache.get(cache_key)` | `report_progress` | +| `care/emr/reports/report_utils.py` | 39 | `clear_lock` → `cache.delete(cache_key)` | `report_progress` | + +**verified** Despite the `_lock` naming, these do **not** use `nx`. They store an +integer progress percentage under the key +`f"report_generation_lock:{key}"` (`report_utils.py:28, 33, 38`), written at +`report_generation.py:35` (10) and `:46` (30), and cleared in a `finally` block +at `:78`. + +**verified** Default TTL is `LOCK_DURATION = 2 * 60` = 120 s +(`report_utils.py:21`), independent of `settings.LOCK_TIMEOUT`. + +**verified** Read back by the API at +`care/emr/api/viewsets/report/report_upload.py:140-145`, which returns HTTP 409 +with the current progress rather than starting a second generation. + +**verified** Because this is a plain `set` and not an `nx` guard, two concurrent +requests can both pass the 409 check before either writes progress. The +"lock" is advisory only. + +**Viable backends:** + +- `Redis-compatible backend` — works today. +- `PostgreSQL database cache` — **viable.** Plain get/set/delete. +- `explicit PostgreSQL model` — **viable and arguably better**, since the progress + belongs to a `ReportUpload` row that already exists. +- `LocMem` — **not viable.** The writer is a Celery worker and the reader is an + API process. They are different processes and, on Cloud Run, different + containers. + +**inferred** This is the clearest example of state that must be shared across the +API/worker boundary. Under Cloud Tasks, the writer becomes a separate Cloud Run +request, so the requirement stands unchanged. + +### 4.8 Rate limiting + +| File | Line | Symbol | Role | +| --- | --- | --- | --- | +| `config/ratelimit.py` | 3 | `from django_ratelimit.core import is_ratelimited` | `rate_limit` | +| `config/ratelimit.py` | 8-9 | `get_ratelimit_key` — returns the constant `"ratelimit"` | `rate_limit` | +| `config/ratelimit.py` | 47 | `is_ratelimited(...)` | `rate_limit` | +| `config/auth_views.py` | 54 | login rate limit | `rate_limit` | +| `care/emr/utils/mfa.py` | 37, 43 | MFA login, by IP and by user | `rate_limit` | +| `care/users/reset_password_views.py` | 60, 114, 169 | password reset, `10/h` | `rate_limit` | + +**verified** `django_ratelimit` is in `INSTALLED_APPS` at +`config/settings/base.py:126`, pinned `==4.1.0` in `Pipfile`. + +**verified** `django_ratelimit` uses the Django cache API. It does **not** require +Redis specifically — but its counter increments are only atomic if the backend +implements atomic `incr`. + +**Viable backends:** + +- `Redis-compatible backend` — works today, atomic `INCR`. +- `PostgreSQL database cache` — **viable but weaker.** `DatabaseCache.incr` is not + atomic against concurrent writers in the way Redis `INCR` is; limits can be + overshot under concurrency. **inferred** acceptable for + password-reset throttling, questionable for security-sensitive MFA limits at + `care/emr/utils/mfa.py:37, 43`. +- `LocMem` — **not viable.** Per-instance counters mean the effective limit is + multiplied by the instance count. On autoscaling Cloud Run this defeats the + control entirely. + +**verified** `config/settings/test.py:58` silences `django_ratelimit.E003`, the +system check that warns when the cache backend is unsuitable for rate limiting. + +### 4.9 Health checks + +| File | Line | Symbol | Role | +| --- | --- | --- | --- | +| `config/settings/base.py` | 457 | `DjangoCacheHealthCheck("Cache", ..., connection_name="default")` | `health_check` | +| `config/settings/base.py` | 458-466 | `DjangoCeleryQueueLengthHealthCheck(..., broker=REDIS_URL, ...)` | `health_check` + `direct_redis` | + +**verified** `DjangoCeleryQueueLengthHealthCheck` is constructed with +`broker=REDIS_URL` (`base.py:461`) and `queue_name="celery"` (`:462`). It connects +to Redis directly to measure queue depth. + +**verified** Thresholds: `info_length=50` (`:463`), `warning_length=0` (`:464`, +commented "this skips the 300 status code"), `alert_length=200` (`:465`). + +**inferred** Under Cloud Tasks this check has no meaning — there is no Redis queue +to measure. It must be removed or replaced with a Cloud Tasks queue-depth probe, +otherwise the health endpoint reports unhealthy in the target runtime. + +### 4.10 Sessions + +**verified** `django.contrib.sessions` is in `INSTALLED_APPS` +(`config/settings/base.py:114`). + +**unknown** No `SESSION_ENGINE` setting appears in `base.py`. **inferred** Django's +default is `django.contrib.sessions.backends.db`, which stores sessions in +PostgreSQL, not Redis. Sessions are therefore **not** a Redis dependency. +Recorded as low-confidence in `unresolved-items.md` §8. + +### 4.11 Not cache uses — name collisions + +**verified** The following matched a `cache.` grep but are **local variables**, not +the Django cache. Listed so they are not mistaken for cache call sites: + +| File | Lines | +| --- | --- | +| `care/security/authorization/token.py` | 104, 105, 143, 144 | +| `care/security/authorization/scheduling.py` | 57, 58, 96, 97 | +| `care/security/authorization/booking.py` | 54, 55, 114, 115 | +| `care/emr/models/questionnaire.py` | 114, 115, 138, 139 | + +**verified** In each case `cache` is a local list built with `.extend(...)` and +`.append(...)` from `organization__parent_cache`. There is no `django.core.cache` +import in these files. + +**verified** `care/facility/models/facility.py` is a special case: it **does** +import the Django cache at line 4, but never calls it. At line 226 it binds a +local `cache = []` inside `sync_cache`, shadowing the import. The import is dead +and the name collides. Not a cache call site. + +--- + +## 5. Roll-up by role + +| Role | Sites | Redis strictly required? | +| --- | --- | --- | +| `celery_broker` | 1 | replaced by Cloud Tasks | +| `celery_result_backend` | 1 | **no** — no consumer exists | +| `distributed_lock` | 4 | **yes**, unless replaced by an explicit PostgreSQL mechanism | +| `direct_redis` | 7 | **yes**, unless replaced by an explicit PostgreSQL model | +| `shared_cache` | 12 | no — PostgreSQL viable; `delete_pattern` needs rework | +| `performance_cache` | 27 | no — PostgreSQL or LocMem viable | +| `transient_state` | 4 | no — PostgreSQL viable; LocMem unsafe | +| `report_progress` | 3 | no — PostgreSQL viable; LocMem unsafe | +| `rate_limit` | 6 invocations + 3 definition lines in `config/ratelimit.py` | no — PostgreSQL viable but weaker; LocMem unsafe | +| `health_check` | 2 | one is Redis-specific and must be replaced | +| `session` | 0 | **no** — DB-backed by default (inferred) | + +**verified** Non-test call-site totals, counted mechanically: + +| Category | Count | How counted | +| --- | --- | --- | +| Django cache API calls (`get`/`set`/`delete`/`add`/`get_or_set`/`delete_many`/`delete_pattern`/`clear`/`incr`) | **52** | grep over `care/` and `config/`, excluding tests and the §4.11 name collisions | +| Raw Redis client operations | **7** | `care/emr/models/valueset.py` §4.3 | +| `ratelimit(...)` invocations | **6** | `auth_views.py:54`; `mfa.py:37, 43`; `reset_password_views.py:60, 114, 169` | +| Redis-dependent settings entries | **4** | `base.py:421, 423, 457, 458-466` | +| **Total** | **69** | | + +Per-file breakdown of the 52 Django cache API calls: + +```text +6 care/security/models/role.py 4 care/utils/lock.py +6 care/emr/models/facility_config.py 4 care/emr/resources/base.py +5 care/users/api/viewsets/plug_config.py 4 care/emr/api/viewsets/favorites.py +5 care/emr/api/viewsets/valueset.py 3 config/auth_views.py +4 care/utils/models/base.py 3 care/emr/reports/report_utils.py +2 care/emr/resources/favorites/filters.py 2 care/emr/models/favorites.py +2 care/emr/fhir/resources/base.py 1 config/authentication.py +1 care/emr/resources/tag/cache_invalidation.py +``` + +Test-file cache calls are excluded and listed in §6. + +--- + +## 6. Test-only cache sites + +**verified** — excluded from the 67: + +| File | Lines | +| --- | --- | +| `care/utils/tests/base.py` | 59, 79, 80 | +| `care/emr/tests/test_favorites_api.py` | 21, 22, 52, 53, 89, 110, 153, 169, 182, 227, 233, 243, 251 | +| `care/emr/tests/test_valueset_api.py` | 23, 52, 462 | +| `care/emr/tests/test_reset_password_api.py` | 24 | + +--- + +## 7. Assessment against the "keep Redis optional" goal + +**verified blockers**, in the order they must be resolved: + +1. `care/utils/lock.py:18, 44` — `nx=True`. Needs a real PostgreSQL locking + primitive. **The shim in `config/caches.py` is not one**; it disables locking. +2. `care/emr/resources/base.py:313, 315` — `delete_pattern`. Needs explicit key + tracking or a custom backend. +3. `care/emr/models/valueset.py:77-122` — raw Redis LIST operations. Needs a + schema addition. +4. `config/settings/base.py:458-466` — Redis-broker health check. Needs removal + or replacement. +5. `config/settings/test.py:45-46` — the test suite itself points at Redis. Making + Redis optional in production without addressing this leaves tests unable to + exercise the PostgreSQL path. + +**inferred** Items 1 and 3 are schema/design work, not configuration. Item 2 is a +contained refactor. Items 4 and 5 are configuration. The claim "Redis is optional" +is not supportable until at least 1, 2 and 3 are done — and item 1 is a +correctness hazard that fails silently rather than loudly. diff --git a/docs/xii/architecture/inventory/frontend-file-flow.md b/docs/xii/architecture/inventory/frontend-file-flow.md new file mode 100644 index 0000000000..06b8b56f8b --- /dev/null +++ b/docs/xii/architecture/inventory/frontend-file-flow.md @@ -0,0 +1,637 @@ +--- +title: Frontend File-Flow Inventory +document: inventory/frontend-file-flow +version: 0.2.0 +status: Draft +phase: 1 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +reviewed: 2026-08-06 +--- + +# Frontend File-Flow Inventory + +The file-related API contract as it exists today, and the exact changes required +to route all file traffic through Django. + +This document is layered, and later sections supersede earlier ones: + +- **Sections 1-10** are the Phase 0 baseline — the contract *before* any storage + or transport work. IS-01 changed no API and no response field, so they stayed + accurate through it. They describe presigned URLs and a base64 upload, neither + of which still exists; read them as history. +- **Section 11** records what IS-01 moved (persistence onto Django Storage, + transport onto CARE routes) and corrects the file and line references it + relocated. Its "IS-02" entries were open at the time of writing. +- **Section 12** records what ES-02 delivered and is **the current contract**. + Where 11 and 12 disagree, 12 is current. + +Evidence labels: **verified** / **inferred** / **unknown**. + +**Scope note:** this repository is the Django backend only. The CARE frontend +lives in a separate repository. Everything below is derived from the backend's +routes, serializers and tests. Frontend behavior is **inferred** from the response +contract, never observed. + +--- + +## 1. Routes + +**verified** `config/api_router.py`: + +| Line | Registration | Base path | +| --- | --- | --- | +| 122 | `router.register("files", FileUploadViewSet, basename="files")` | `/api/v1/files/` | +| 500 | `router.register("template_reports", ReportUploadViewSet, basename="template-reports")` | `/api/v1/template_reports/` | + +**verified** `FileUploadViewSet` (`care/emr/api/viewsets/file_upload.py:119-121`) +mixes in `EMRCreateMixin`, `EMRRetrieveMixin`, `EMRUpdateMixin`, `EMRListMixin`. +There is **no destroy mixin** — files are archived, not deleted. + +--- + +## 2. Current upload flow — two paths coexist + +### 2.1 Path A: presigned PUT (the default) + +**verified** Three steps: + +| Step | Endpoint | Handler | What the client gets | +| --- | --- | --- | --- | +| 1. Initiate | `POST /api/v1/files/` | `EMRCreateMixin` → `FileUploadCreateSpec` | `FileUploadRetrieveSpec` including **`signed_url`** | +| 2. Upload | **direct to object storage** | none — browser PUTs to the bucket | — | +| 3. Complete | `POST /api/v1/files/{external_id}/mark_upload_completed/` | `file_upload.py:177-184` | `FileUploadListSpec` | + +**verified** Step 1 returns a write URL because +`FileUploadCreateSpec.perform_extra_deserialization` sets +`obj._just_created = True` (`care/emr/resources/file_upload/spec.py:51`), and +`FileUploadRetrieveSpec.perform_extra_serialization` branches on that flag: + +```python +# care/emr/resources/file_upload/spec.py:110-117 +@classmethod +def perform_extra_serialization(cls, mapping, obj): + super().perform_extra_serialization(mapping, obj) + if getattr(obj, "_just_created", False): + # Calculate Write URL and return it + mapping["signed_url"] = obj.files_manager.signed_url(obj) # line 115 + else: + mapping["read_signed_url"] = obj.files_manager.read_signed_url(obj) # line 117 +``` + +**verified** `signed_url` is a presigned **`put_object`** URL +(`care/emr/utils/file_manager.py:46-47`) with a 3600 s default expiry +(`file_manager.py:35`). + +**verified** Step 3 sets `upload_completed = True` (`file_upload.py:181`). This +matters because `get_queryset` filters list results on `upload_completed=True` +(`file_upload.py:169`) — a file never marked complete is invisible to listing. + +**verified** Nothing verifies that an object actually exists in the bucket before +`mark_upload_completed` flips the flag. The endpoint trusts the client. + +### 2.2 Path B: base64 through Django (already exists) + +**verified** `POST /api/v1/files/upload-file/` +(`care/emr/api/viewsets/file_upload.py:213-270`, `url_path="upload-file"`). + +**verified** Request body fields, read at `file_upload.py:215-216, 246-253`: + +```text +original_name (required) file_upload.py:215 +file_data (required) base64 string, file_upload.py:216 +name file_upload.py:248 +associating_id file_upload.py:249 +file_type file_upload.py:250 +file_category file_upload.py:251 +``` + +**verified** `mime_type` is **not** taken from the client — it is sniffed server +side with `magic.from_buffer(file_content[:2048], mime=True)` +(`file_upload.py:237`) and checked against `settings.ALLOWED_MIME_TYPES` +(`file_upload.py:242-244`). + +**verified** Response is `FileUploadRetrieveSpec` (`file_upload.py:270`). Because +`file_upload.py:257` sets `_just_created = False`, the response contains +**`read_signed_url`**, not `signed_url` — so even this Django-proxied upload +hands back a direct object-storage read URL. + +**verified** This path already satisfies "uploads pass through Django". It does +**not** satisfy "downloads pass through Django". + +**verified** No `mark_upload_completed` call is needed — `file_upload.py:263` +sets the flag inline. + +--- + +## 3. Current download flow + +**verified** There is **no download endpoint**. Downloads happen entirely against +object storage. + +| Response field | Produced at | Content | +| --- | --- | --- | +| `read_signed_url` | `care/emr/resources/file_upload/spec.py:117` | presigned `get_object` URL | +| `read_signed_url` | `care/emr/resources/report/report_upload/spec.py:54` | presigned `get_object` URL | + +**verified** `read_signed_url` (`care/emr/utils/file_manager.py:52-69`) sets +`ResponseContentDisposition` (`:66`) using a MIME allowlist: + +```python +# care/emr/utils/file_manager.py:11-20, 56-59 +SAFE_INLINE_FORMATS = { + "image/jpeg", "image/png", "image/gif", "image/webp", + "image/tiff", "image/bmp", "image/x-icon", "application/pdf", +} +... +mime_type = file_obj.meta.get("mime_type") +content_disposition = "inline" if mime_type in SAFE_INLINE_FORMATS else "attachment" +``` + +**inferred** The frontend renders images and PDFs inline and downloads everything +else, relying on the storage provider to honor `ResponseContentDisposition`. Any +Django-served replacement must set the same `Content-Disposition` header or the +browser behavior changes for every non-image attachment. + +**verified** Expiry is 3600 s (`file_manager.py:52`). + +--- + +## 4. Unsigned public URLs — cover images and avatars + +**verified** A third file flow exists that uses neither presigned URLs nor +`S3FilesManager`. + +| Endpoint | Handler | Line | +| --- | --- | --- | +| `POST/DELETE /api/v1/facility/{external_id}/cover_image/` | `FacilityViewSet.cover_image` | `care/emr/api/viewsets/facility.py:119-121` | +| user profile picture | `care/emr/api/viewsets/user.py:43-48` | serializer fields | + +**verified** These accept a **multipart `ImageField`** +(`facility.py:39-42`, `user.py:43-46`) — a genuine file upload through Django, +not base64, not presigned. The action is explicitly bound to `MultiPartParser` +at `facility.py:119` via +`@method_decorator(parser_classes([MultiPartParser]))`. + +**inferred** This is the closest existing precedent in the codebase for the +streaming multipart upload that change U2 (§9.1) requires — worth reusing rather +than designing fresh. + +**verified** Response fields are `read_cover_image_url` +(`facility.py:44`) and `read_profile_picture_url` (`user.py:48`), both +`serializers.URLField(read_only=True)`. + +**verified** Those URLs are built by string concatenation against the bucket: + +```python +# care/facility/models/facility.py:207-212 +def read_cover_image_url(self): + if self.cover_image_url: + if settings.FACILITY_CDN: + return f"{settings.FACILITY_CDN}/{self.cover_image_url}" + return f"{settings.FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT}/{settings.FACILITY_S3_BUCKET}/{self.cover_image_url}" + return None +``` + +**verified** `care/users/models.py:202-207` is identical in shape. + +**verified** These objects are written with `ACL: public-read` when +`settings.BUCKET_HAS_FINE_ACL` is true (`care/utils/file_uploads/cover_image.py:49-51`). + +**inferred** Cover images and avatars are therefore **public, unauthenticated +objects**. Unlike patient files, no signature gates them. A GCS bucket with +uniform bucket-level access breaks both the write (ACL rejected) and the read +(object not public). + +--- + +## 5. Serializers and response contract + +**verified** Pydantic specs, not DRF serializers, for files and reports: + +| Spec | File | Fields carrying storage URLs | +| --- | --- | --- | +| `FileUploadRetrieveSpec` | `care/emr/resources/file_upload/spec.py:105-117` | `signed_url` (`:106`), `read_signed_url` (`:107`) | +| `FileUploadListSpec` | `care/emr/resources/file_upload/spec.py:76-102` | **none** | +| `ReportUploadRetrieveSpec` | `care/emr/resources/report/report_upload/spec.py:43-54` | `signed_url` (`:44`), `read_signed_url` (`:45`) | +| `ReportUploadListSpec` | `care/emr/resources/report/report_upload/spec.py:19-40` | **none** | + +**verified** Both URL fields are `str | None = None` — optional and mutually +exclusive in practice, because the `_just_created` branch sets exactly one. + +**verified** `FileUploadRetrieveSpec` also exposes `internal_name` +(`spec.py:108`), carrying an in-source comment: +`# Not sure if this needs to be returned`. `internal_name` is the storage object +key (`care/emr/models/file_upload.py:45-49`). **inferred** Leaking the object key +is low risk while the bucket is private, but it is unnecessary surface. + +**verified** DRF serializers are used for the cover-image flow only +(`facility.py:39-49`, `user.py:43-48`). + +--- + +## 6. OpenAPI schema + +**verified** `drf-spectacular` is the schema generator (`Pipfile`, +`drf-spectacular = "==0.29.0"`). + +**verified** Explicit schema annotations on the file endpoints: + +| Endpoint | Annotation | Line | +| --- | --- | --- | +| `mark_upload_completed` | `@extend_schema(responses={200: FileUploadListSpec})` | `file_upload.py:176` | +| `archive` | `@extend_schema(request=ArchiveRequestSpec, responses={200: FileUploadListSpec})` | `file_upload.py:189-192` | +| report `archive` | `@extend_schema(request=ArchiveRequestSpec, responses={200: ReportUploadListSpec})` | `report/report_upload.py:162` | + +**verified** `upload_file` at `file_upload.py:213` has **no** `@extend_schema` +decorator. Its request body — `original_name` and base64 `file_data` — is read +directly from `request.data` and is therefore **absent from the generated +OpenAPI schema**. Any client generated from the schema cannot discover this +endpoint's contract. Recorded in `unresolved-items.md` §9. + +**verified** A dedicated locmem cache is reserved for schema generation: +`"swagger_cache"` at `config/settings/base.py:96-99`. + +--- + +## 7. Tests covering the contract + +**verified** `care/emr/tests/test_file_upload_api.py`: + +| Line | What it asserts | +| --- | --- | +| 19 | `@override_settings(FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT=settings.BUCKET_ENDPOINT)` on the test class | +| 77 | `response.data["signed_url"]` present on create | +| 102 | `response.data["read_signed_url"]` present on retrieve | +| 137 | `read_signed_url` present | +| 165 | `signed_url` present | +| 177 | `cleanup_incomplete_file_uploads.delay()` | +| 180 | `file_obj.files_manager.get_object(file_obj)` raises after cleanup | + +**verified** The `@override_settings` at line 19 exists because signed URLs are +generated against the **external** endpoint while the test process reaches +storage on the **internal** one. This is the `external=True` branch in +`care/utils/csp/config.py:73` propagating into tests. + +**verified** These tests assert on the **presence of the URL fields themselves**. +Removing `signed_url` / `read_signed_url` from the response breaks the suite at +lines 77, 102, 137 and 165 — the tests are coupled to the presigned-URL design, +not merely to file behavior. + +**verified** The suite requires reachable object storage: line 180 performs a real +`get_object` and expects `ClientError` (imported at `test_file_upload_api.py:6`). + +--- + +## 8. Frontend repository references + +**verified** No frontend source exists in this repository. + +**verified** Grep for `care_fe`, `ohcnetwork/care_fe` and similar across the +repository returns no code reference. The only cross-repository coupling found is +in CI: `.github/workflows/reusable-test.yml:98-109` uploads a database dump named +`care-db-dump` with the comment *"Upload dummy db as artifact so it can be used +to speed up frontend tests"*. + +**unknown** Which frontend components consume `signed_url` / `read_signed_url`, +and whether any perform a direct browser PUT versus using the base64 endpoint. +This cannot be determined here. Recorded in `unresolved-items.md` §10. + +**inferred** Because Path B (`upload-file`) exists and is routed, at least one +client is expected to use it. But since it is absent from the OpenAPI schema +(§6), it was likely added for a specific non-browser caller. Unconfirmed. + +--- + +## 9. Exact API changes required to route all traffic through Django + +The stated goal: all uploads and downloads pass through Django; no direct +browser-to-bucket transfer. + +### 9.1 Upload + +| # | Change | Location | +| --- | --- | --- | +| U1 | Stop returning `signed_url`. Remove the `_just_created` write branch. | `care/emr/resources/file_upload/spec.py:113-115`; `care/emr/resources/report/report_upload/spec.py:51-52` | +| U2 | Add a multipart upload endpoint that streams to storage rather than buffering base64. | new action on `FileUploadViewSet` | +| U3 | Decide the fate of the existing base64 endpoint — keep for compatibility or deprecate. | `care/emr/api/viewsets/file_upload.py:213-270` | +| U4 | Re-express `mark_upload_completed` as either server-set or a no-op retained for compatibility. | `care/emr/api/viewsets/file_upload.py:177-184` | +| U5 | Remove the presigned write method once no caller remains. | `care/emr/utils/file_manager.py:35-50` | + +**verified** U1 is not cosmetic: `FileUploadRetrieveSpec` is the create response +(`file_upload.py:123-124` sets `pydantic_model` / `pydantic_retrieve_model`), so +removing the field changes the create contract. + +### 9.2 Download + +| # | Change | Location | +| --- | --- | --- | +| D1 | Add an authenticated download route, e.g. `GET /api/v1/files/{external_id}/download/`. | new action on `FileUploadViewSet` | +| D2 | Reproduce the `inline` vs `attachment` decision in a real `Content-Disposition` header. | port `care/emr/utils/file_manager.py:56-59, 66` | +| D3 | Stream the object rather than buffering. `file_contents` (`file_manager.py:90-94`) reads whole bodies. | `care/emr/utils/file_manager.py:90-94` | +| D4 | Apply `file_authorizer` on the download path. | reuse `care/emr/api/viewsets/file_upload.py:38-103` | +| D5 | Stop returning `read_signed_url`; return a Django URL instead. | `spec.py:117`; `report/report_upload/spec.py:54` | +| D6 | Same for the report data point used in report context. | `care/emr/reports/context_builder/data_points/fileupload.py:29` | + +**verified** D4 is not automatic. `file_authorizer` currently runs in +`get_queryset` (`file_upload.py:157-173`) and in the create/update authorize +hooks. A new download action must call it explicitly, or it will serve any file +to any authenticated user. + +**verified** D6 matters because +`care/emr/reports/context_builder/data_points/fileupload.py:29` embeds +`read_signed_url` into **generated report content**. A 1-hour signed URL baked +into a stored PDF expires; a Django URL does not. This is a behavior improvement, +but it changes what is rendered inside reports. + +### 9.3 Cover images and avatars + +| # | Change | Location | +| --- | --- | --- | +| C1 | Decide whether these objects stay public. | `care/utils/file_uploads/cover_image.py:49-51` | +| C2 | If private, replace both URL builders with Django routes. | `care/facility/models/facility.py:207-212`; `care/users/models.py:202-207` | +| C3 | Resolve the ACL call under uniform bucket-level access. | `care/utils/file_uploads/cover_image.py:49-51` | + +### 9.4 Tests + +| # | Change | Location | +| --- | --- | --- | +| T1 | Rewrite the four URL-presence assertions. | `care/emr/tests/test_file_upload_api.py:77, 102, 137, 165` | +| T2 | Reconsider the external-endpoint override, which exists only for presigned URLs. | `care/emr/tests/test_file_upload_api.py:19` | + +### 9.5 Schema + +| # | Change | Location | +| --- | --- | --- | +| S1 | Add `@extend_schema` to the upload endpoint so it appears in OpenAPI. | `care/emr/api/viewsets/file_upload.py:213` | +| S2 | Regenerate and diff the schema; `signed_url` / `read_signed_url` disappear from two response models. | — | + +--- + +## 10. Summary of contract impact + +**verified** Response fields removed or repurposed: **4** +(`signed_url` and `read_signed_url` on both `FileUploadRetrieveSpec` and +`ReportUploadRetrieveSpec`). + +**verified** Endpoints added: **at least 2** (upload, download). Possibly 2 more +if cover images and avatars stop being public. + +**verified** Endpoints whose meaning changes: **1** +(`mark_upload_completed`, `file_upload.py:177-184`). + +**verified** Tests requiring rewrite: **4 assertions in 1 file**. + +**inferred** The frontend must change in lockstep — this is a **breaking API +change**, not an additive one, unless both URL fields are retained as Django URLs +under their existing names. Retaining the names is the lower-risk path and would +let the backend migrate before the frontend, but it leaves the misleading +`signed_url` naming in place. + +--- + +## 11. After IS-01 + +Recorded 2026-08-06. **No route, request field or response field changed.** The +transport contract in §§1-10 is intact; only what happens beneath it moved. + +### 11.1 Relocated references + +The code these sections point at moved. Substitute as follows: + +| §§ referring to | Now lives at | +| --- | --- | +| `care/emr/utils/file_manager.py:35-50` (`signed_url`) | `care/emr/utils/legacy_signed_urls.py:signed_url` | +| `care/emr/utils/file_manager.py:52-69` (`read_signed_url`) | `care/emr/utils/legacy_signed_urls.py:read_signed_url` | +| `care/emr/utils/file_manager.py:11-20` (`SAFE_INLINE_FORMATS`) | `care/emr/utils/legacy_signed_urls.py:SAFE_INLINE_FORMATS` | +| `obj.files_manager.signed_url(obj)` in both specs | `legacy_signed_urls.signed_url(obj)` | +| `obj.files_manager.read_signed_url(obj)` in both specs and the data point | `legacy_signed_urls.read_signed_url(obj)` | +| `care/emr/utils/file_manager.py:90-94` (`file_contents`, D3) | same file, now `Storage.open().read()` | + +**verified** §7's statement that the suite expects `ClientError` at +`test_file_upload_api.py:180` is superseded: the test now asserts the object does +not exist and that opening it raises `FileNotFoundError`. The four URL-presence +assertions at lines 77, 102, 137 and 165 are **unchanged**, so T1 in §9.4 is +still entirely IS-02 work. + +**verified** The `@override_settings` at `test_file_upload_api.py:19` is also +unchanged, so T2 remains IS-02 work. It is still required, because signed URLs +are still generated against the external endpoint. + +### 11.2 What now uses Django Storage + +| Flow | Transport | Persistence after IS-01 | +| --- | --- | --- | +| Path A, presigned PUT | unchanged — browser writes directly to the bucket | none: Django never sees the bytes | +| Path A, step 1 `signed_url` | unchanged | `legacy_signed_urls`, boto3, S3 only | +| Path B, base64 `upload-file` | unchanged — still base64, still buffered | **`Storage.save()`** via `storages["patient"]` | +| Download `read_signed_url` | unchanged | `legacy_signed_urls`, boto3, S3 only | +| Report generation | not a browser flow | **`Storage.save()`** via `storages["report"]` | +| Cover image / avatar upload | unchanged multipart | **`Storage.save()`** via `storages["facility"]`, now streamed rather than buffered | +| Cover image / avatar URLs | unchanged string concatenation | not storage persistence; still bypasses the alias | +| Cleanup task | not a browser flow | **`Storage.delete()`** | + +### 11.3 Flows removed + +**Signed upload (browser-to-bucket PUT) — gone.** The `signed_url` field and the +`_just_created` write branch are deleted from both retrieve specs. Clients upload +through `POST /api/v1/files/upload-file/`. + +**Signed download (browser-from-bucket GET) — gone.** `read_signed_url` is +replaced by `download_url`, a CARE route, on both specs and in the report data +point. + +**Unsigned public bucket URLs — gone.** `Facility.read_cover_image_url` and +`User.read_profile_picture_url` now reverse to CARE asset routes instead of +concatenating `{endpoint}/{bucket}/{key}`. `FACILITY_CDN` and +`BUCKET_HAS_FINE_ACL` are deleted; no object carries a public ACL. + +**Base64 upload — gone (ES-02).** `POST /api/v1/files/upload-file/` now accepts +`multipart/form-data`. See §12. + +### 11.4 Effect on the §9 change list + +| Item | Status | +| --- | --- | +| U1 stop returning `signed_url` | **done** | +| U2 multipart streaming upload | **IS-02** — the only upload item left | +| U3 fate of the base64 endpoint | **IS-02** | +| U4 `mark_upload_completed` | **IS-02.** Still client-driven and still unverified against storage. Now bounded rather than fixed: a client can no longer write to the bucket, and an incomplete row advertises no `download_url` and refuses `download` (§11.7), so a falsely-completed row exposes nothing — it only becomes listable | +| U5 remove the presigned write | **done** | +| D1 authenticated download route | **done** | +| D2 `Content-Disposition` | **done** — inline/attachment preserved in `file_download.py` | +| D3 stream rather than buffer | **done** — `FileResponse` streams; `file_contents` is an opt-in with no production caller | +| D4 authorize the download path | **done** — files reuse `get_queryset`'s `file_authorizer`; reports authorize explicitly | +| D5 stop returning `read_signed_url` | **done** | +| D6 report data point | **done** — embeds a CARE route, which also fixes the expiring-URL-inside-a-stored-PDF problem | +| C1 do these objects stay public | **decided: served by CARE, bucket private.** The routes stay anonymous, so visibility is unchanged | +| C2 replace both URL builders | **done** | +| C3 ACL under uniform bucket-level access | **done** — no ACL is ever set | +| T1, T2 test rewrites | **done** | +| S1 schema for the upload endpoint | **IS-02** | +| S2 regenerate and diff the schema | **partly** — `signed_url` and `read_signed_url` are gone from both response models; `download_url` is added | + +**Every row above marked IS-02 is now closed by ES-02** — U2, U3 and S1 outright, +U4 by deprecation rather than removal. The table is left as the IS-01 record; +§12 is the current state. + +### 11.5 Contract impact, as delivered + +**Breaking for the frontend.** Response fields removed: `signed_url` and +`read_signed_url` on both `FileUploadRetrieveSpec` and +`ReportUploadRetrieveSpec`. Added: `download_url` on both. + +`read_cover_image_url` and `profile_picture_url` keep their names but now hold a +CARE path rather than an absolute bucket URL. + +All four are **relative** paths (`/api/v1/...`). The previous values were +absolute. A client that concatenated them onto an API base will keep working; one +that used them as absolute URLs must prepend the API origin. + +New endpoints: 4 (two downloads, two public assets). `mark_upload_completed` +retains its meaning for now. + +`download_url` is `null` while `upload_completed` is false — see §11.7. + +### 11.6 GCS is no longer blocked on transport + +**verified** With the signed-URL path gone, selecting `CARE_STORAGE_BACKEND=gcs` +now yields a fully working file surface: persistence and transport both go +through Django Storage. The earlier constraint — that IS-02 was a prerequisite +for any real GCS deployment — no longer holds. + +**verified** One provider-specific reference remains outside transport: +`report_generation.py` retries on `botocore`'s `ClientError`, which will not fire +under `gcs`. See `storage-call-sites.md` §11.7. + +### 11.7 An incomplete row is not downloadable + +**verified** A `FileUpload` row exists before its bytes do: `POST /api/v1/files/` +creates it with `upload_completed=False`, and only `mark_upload_completed` flips +the flag. The detail queryset does not filter on that flag (`file_upload.py`, +`get_queryset`), so such a row is retrievable. + +`FileUploadRetrieveSpec` therefore returns `download_url: null` until the upload +completes, and the `download` action raises `NotFound` for an incomplete row +rather than reaching storage. Previously it advertised a route that could only +404 — a provider-neutral 404, but still a URL for an object that was never +written. + +This bounds the vestigial create/complete pair without removing it: a client that +calls `mark_upload_completed` without uploading makes the row listable, but it +exposes nothing, because there is no URL and no route that resolves. See §12.5. + +--- + +## 12. After ES-02 — file transport + +Recorded 2026-08-07. ADR-0002 is now fully implemented. + +### 12.1 Flow status + +| Flow | Status | +| --- | --- | +| Signed upload (browser → bucket) | **removed** (ES-01) | +| Signed download (browser ← bucket) | **removed** (ES-01) | +| Unsigned public bucket URLs | **removed** (ES-01) | +| Base64 upload | **removed** (ES-02) | +| Multipart upload | **implemented** (ES-02) | +| Server-mediated download | **implemented** (ES-01) | + +Every byte in and out of object storage now passes through CARE, and no client +holds a storage-provider URL of any kind. + +### 12.2 Upload contract + +```http +POST /api/v1/files/upload-file/ +Content-Type: multipart/form-data +``` + +| Part | Required | Notes | +| --- | --- | --- | +| `file` | yes | the binary file part | +| `name` | yes | display name | +| `file_type` | yes | `FileTypeChoices` | +| `file_category` | yes | `FileCategoryChoices` | +| `associating_id` | yes | external id of the owning object | +| `original_name` | no | defaults to the uploaded part's filename | + +**Removed:** `file_data`. There is no fallback — a base64 body is now rejected. + +**verified** `original_name` is newly optional. Multipart supplies a filename; +base64-over-JSON could not, which is why it used to be mandatory. + +Response is unchanged (`FileUploadRetrieveSpec`), including the relative +`download_url`. + +### 12.3 Client migration + +The frontend is a separate repository, so the contract change is recorded here +rather than implemented. + +```js +const body = new FormData(); +body.append("file", fileObject); // was: file_data, base64 string +body.append("name", name); +body.append("file_type", fileType); +body.append("file_category", fileCategory); +body.append("associating_id", associatingId); + +await fetch("/api/v1/files/upload-file/", { method: "POST", body }); +// Do not set Content-Type; the browser sets the multipart boundary. +``` + +**Breaking.** A client still sending `file_data` receives 400. + +### 12.4 Memory and size + +**verified** The file is never fully materialised by CARE. Django's upload +handlers decide the representation before the view runs; size comes from +`UploadedFile.size`; MIME sniffing reads only the leading 2048 bytes; and the +`UploadedFile` is handed straight to `Storage.save()`. + +| Setting | Default | Effect | +| --- | --- | --- | +| `MAX_FILE_UPLOAD_SIZE` | 5 (MB) | CARE rejects anything larger, before any write | +| `FILE_UPLOAD_MAX_MEMORY_SIZE` | 2621440 (2.5 MB) | above this Django spools to a `TemporaryUploadedFile` | +| `DATA_UPLOAD_MAX_MEMORY_SIZE` | 2621440 (2.5 MB) | bounds the non-file part only; multipart file parts are exempt | + +The latter two were previously Django's implicit defaults; ES-02 states them +explicitly at the same values, so behaviour is unchanged. + +**verified, and worth recording:** base64 inflated a 5 MB file to roughly 6.7 MB +of JSON body, which exceeds `DATA_UPLOAD_MAX_MEMORY_SIZE`. The old endpoint +could not actually accept a file at its own documented limit over JSON. +Multipart removes that ceiling because file parts are exempt. + +### 12.5 Unchanged by ES-02 + +**verified** Cover images and avatars already used ordinary multipart +`ImageField` uploads (`FacilityImageUploadSerializer`, +`UserImageUploadSerializer`) and never went through the base64 endpoint. They +are untouched, as ES-02 §26 permits. + +**verified** Authorization, object naming, alias selection and persistence are +unchanged: `authorize_create` → `file_authorizer`, `get_storage_name`, the +`patient` alias, and `Storage.save()`. + +**verified** `POST /api/v1/files/` and `mark_upload_completed` still exist. They +were the first and third steps of the presigned flow. Nothing writes to storage +between them any more, so the pair is now vestigial — a row created this way can +only be completed by a client asserting it, and the assertion is never checked +against storage. + +**Recommend deprecating the pair rather than leaving it as an accepted state.** +The multipart endpoint creates the row, writes the object and sets the flag in +one authorized request, so nothing needs the two-step form any more. Leaving it +in place keeps an endpoint whose entire function is to let a client assert +something the server could verify — and after ES-02 there is no flow in which +that assertion is the truth. + +It is bounded rather than dangerous: `download_url` is `null` and `download` +returns 404 while the object is missing (§11.7), so a falsely-completed row +exposes nothing and merely appears in listings. Out of scope for ES-02. Whoever +owns the file API next SHOULD either set the flag server-side, verify the object +exists before setting it, or remove both endpoints. diff --git a/docs/xii/architecture/inventory/plugin-impact.md b/docs/xii/architecture/inventory/plugin-impact.md new file mode 100644 index 0000000000..79a15bf1d5 --- /dev/null +++ b/docs/xii/architecture/inventory/plugin-impact.md @@ -0,0 +1,267 @@ +--- +title: Plugin Impact Inventory +document: inventory/plugin-impact +version: 0.2.0 +status: Draft +phase: 1 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +reviewed: 2026-08-07 +--- + +# Plugin Impact Inventory + +How CARE loads plugins, what is bundled at this commit, and what can and cannot +be determined about plugin compatibility from this repository alone. + +Evidence labels: **verified** / **inferred** / **unknown**. + +--- + +## 1. Headline + +**verified** **No plugins are bundled at this commit.** `plug_config.py:4` is: + +```python +plugs = [] +``` + +**verified** Therefore `PLUGIN_APPS` is empty, no plugin app is installed, no +plugin URL is routed, and no plugin migration exists **in this configuration**. + +**verified** However, plugins can be injected at build time or runtime **without +touching this repository**, via the `ADDITIONAL_PLUGS` environment variable. So +"no plugins bundled" is not the same as "no plugins in a given deployment". + +--- + +## 2. Loading mechanism + +**verified** Four files implement the whole system: + +| File | Role | +| --- | --- | +| `plug_config.py` | Declares the plug list; instantiates `PlugManager` | +| `plugs/plug.py` | The `Plug` dataclass | +| `plugs/manager.py` | `PlugManager` — install, app list, config aggregation | +| `install_plugins.py` | Build-time entrypoint: `manager.install()` | + +**verified** `plugs/plug.py:5-10` — the `Plug` dataclass: + +```python +@dataclass(slots=True) +class Plug: + name: str + package_name: str + version: str = field(default="@main") + configs: dict = field(default_factory=dict) +``` + +**verified** `version` defaults to `"@main"` (`plug.py:8`) — a **git ref, not a +pinned release**. **inferred** the default installs a moving target; two builds +of the same commit can produce different plugin code. + +### 2.1 Runtime injection + +**verified** `plugs/manager.py:22-29`: + +```python +if additional_plugs := os.getenv("ADDITIONAL_PLUGS"): + try: + for plug in json.loads(additional_plugs): + self.add_plug(Plug(**plug)) + except json.JSONDecodeError: + logger.error("ADDITIONAL_PLUGS is not a valid JSON") +``` + +**verified** `ADDITIONAL_PLUGS` is read from the environment when +`PlugManager.__init__` runs — which happens at +`config/settings/base.py:19` (`from plug_config import manager`), i.e. **at +Django settings import time, in every process**. + +**verified** Malformed JSON is logged and swallowed (`manager.py:27-28`). The +process starts with **zero** plugins rather than failing. **inferred** a typo in +`ADDITIONAL_PLUGS` silently disables every plugin instead of crashing — a +deployment hazard on Cloud Run, where the symptom would be missing endpoints +rather than a failed rollout. + +### 2.2 Installation + +**verified** `plugs/manager.py:31-35`: + +```python +def install(self) -> None: + packages = {f"{x.package_name}{x.version}" for x in self.plugs} + if packages: + subprocess.check_call([sys.executable, "-m", "pip", "install", *packages]) +``` + +**verified** This runs `pip install` in a subprocess. It is invoked from +`install_plugins.py` at **image build time**, `docker/prod.Dockerfile:39`: + +```dockerfile +ARG ADDITIONAL_PLUGS="" +ENV ADDITIONAL_PLUGS=$ADDITIONAL_PLUGS +RUN python3 $APP_HOME/install_plugins.py +``` + +**verified** `ADDITIONAL_PLUGS` is plumbed through compose as a build arg: +`docker-compose.local.yaml:7-8`. + +**verified critical asymmetry:** `ADDITIONAL_PLUGS` is consumed in **two +different phases**: + +| Phase | Consumer | Effect | +| --- | --- | --- | +| Image build | `install_plugins.py` → `manager.install()` | `pip install`s the packages | +| Every process start | `config/settings/base.py:19` → `PlugManager.__init__` | Adds them to `INSTALLED_APPS` | + +**inferred** If the runtime value differs from the build-time value, Django lists +an app in `INSTALLED_APPS` that was never pip-installed, and the process dies at +startup with `ModuleNotFoundError`. On Cloud Run — where the image and the env +vars are configured independently — this is an easy misconfiguration. The +variable must be identical at build and deploy. + +### 2.3 Integration points + +**verified** Exactly three: + +| Integration | Location | Mechanism | +| --- | --- | --- | +| Apps | `config/settings/base.py:142, 149` | `PLUGIN_APPS = manager.get_apps()`; appended to `INSTALLED_APPS` | +| Settings | `config/settings/base.py:146` | `PLUGIN_CONFIGS = manager.get_config()` | +| URLs | `config/urls.py:111-112` | `path(f"api/{plug}/", include(f"{plug}.urls"))` | + +**verified** `config/urls.py:111-112` requires every plugin to expose a +`urls` module. There is no `try`/`except` — a plugin without `urls.py` raises at +import and the process fails to start. + +**verified** `manager.get_config()` (`manager.py:41-49`) returns a +`defaultdict[str, dict]` keyed by plugin name. **unknown** how plugins read it; +no consumer of `PLUGIN_CONFIGS` exists in this repository beyond its definition. + +**verified** `PlugConfig` is also a **database model** with its own viewset +(`care/users/api/viewsets/plug_config.py`), distinct from `PLUGIN_CONFIGS`. +Its `list` action is **unauthenticated** — `get_authenticators` returns `[]` for +`GET` (`plug_config.py:36-39`) — and the response is cached under +`care_plug_viewset_list` (`plug_config.py:14, 17-22`). + +--- + +## 3. Impact assessment per risk category + +Because `plugs = []`, every row below is about what a plugin **could** introduce, +not what one does today. + +| Risk | Determinable here? | Assessment | +| --- | --- | --- | +| Celery tasks | **no** | `app.autodiscover_tasks()` (`config/celery_app.py:18`) scans every app in `INSTALLED_APPS`. Any plugin `tasks.py` is registered automatically and would need a Cloud Tasks route. | +| Redis dependencies | **no** | A plugin can import `django_redis` or call `cache.set(..., nx=True)` freely. Nothing constrains it. | +| Direct S3 / boto3 | **no** | `boto3` is a core dependency, importable by any plugin. | +| Signed URLs | **partly, since ES-01** | CARE no longer generates any. `S3FilesManager` is still importable from `care.emr.utils.file_manager` but exposes no signed-URL method — see §9. A plugin can still construct its own `boto3` client, so the guarantee is CARE's, not the platform's. | +| Custom health checks | **partly** | `HEALTHY_DJANGO` (`config/settings/base.py:453-467`) is a plain list. A plugin cannot append to it through the plug system — no hook exists. **inferred** plugins cannot register health checks. | +| Custom startup behavior | **no** | Standard Django `AppConfig.ready()` is available to any plugin app. | +| Additional migrations | **no** | Plugin apps are ordinary Django apps; their migrations run with `migrate`. Given §2 of `runtime-and-deployment.md`, they would run only in the Celery Beat container. | + +**verified** The health-check row is the only category the plug system +structurally prevents. All others are wide open because plugins are just Django +apps with unrestricted imports. + +--- + +## 4. What cannot be determined + +**unknown**, and not determinable from this repository: + +1. **Which plugins any given deployment runs.** Governed by `ADDITIONAL_PLUGS`, + set outside version control. +2. **Whether known CARE plugins are GCP-compatible.** No plugin source is vendored. +3. **Plugin Celery task shapes** — payloads, retries, idempotency. +4. **Plugin storage usage** — buckets, signed URLs, direct object access. +5. **Plugin Redis usage** — locks, raw clients, pattern deletes. +6. **Plugin migration dependencies** on core CARE tables. +7. **What `PLUGIN_CONFIGS` keys mean**, since no consumer exists here. + +**verified** The repository offers no manifest, lockfile or compatibility matrix +for plugins. `plug_config.py` is the only declaration point and it is empty. + +**inferred** Any statement that "CARE plugins work on GCP" is unsupportable from +this repository. Each deployment's plugin set has to be inventoried separately, +using the same method applied here to core CARE. + +--- + +## 5. Consequences for the GCP migration + +**inferred**, flowing from verified facts above: + +1. **The plugin system is a hole in every other inventory in this directory.** + The storage, task and cache inventories are complete for *core CARE at this + commit*. They are not complete for any deployment with plugins. + +2. **`autodiscover_tasks` means plugin tasks appear without registration** + (`config/celery_app.py:18`). A Cloud Tasks design that enumerates known tasks + by hand will silently drop them. + +3. **`ADDITIONAL_PLUGS` must match between build and deploy** (§2.2), and a JSON + typo disables plugins silently (§2.1). Both deserve a startup assertion. + +4. **`version` defaults to `@main`** (`plugs/plug.py:8`). Reproducible GCP builds + require explicit pins. + +5. **Plugins cannot contribute health checks** (§3), so the Cloud Run health + endpoint stays under core control. + +6. **Plugin migrations inherit the beat-only migration problem** + (`runtime-and-deployment.md` §2). Whatever replaces beat must run them too. + +**Recommendation (inferred):** treat the plugin set as an explicit input to the +GCP design. Before deploying, run this same inventory against each plugin the +target deployment actually installs. Document that set in +`07-configuration-reference.md` rather than assuming the empty default. + +--- + +## 9. Storage API change in ES-01 (deprecation notice) + +Recorded 2026-08-07. + +**verified** `care.emr.utils.file_manager.S3FilesManager` — the one storage +symbol this inventory identified as plugin-reachable — still imports and still +works. It is now a **deprecated** subclass of `FilesManager`. + +What changed: + +| Aspect | Before | After | +| --- | --- | --- | +| Base | own class over `boto3` | `FilesManager`, delegating to Django Storage | +| Constructor argument | `BucketType.PATIENT` | `"PATIENT"` or `"patient"`; the old enum member still works via its `.value` | +| `put_object` / `get_object` / `delete_object` | boto3 calls, returned provider response dicts | Django Storage; return a name, a file object, or `None` | +| `put_object(**kwargs)` | passed provider kwargs through | replaced by an optional `content_type` | +| `file_contents` | `(content_type, bytes)` tuple | `bytes` | +| `delete_object(quiet=...)` | argument accepted | removed; deletion is idempotent | +| `signed_url` / `read_signed_url` | present | **removed** | +| Unknown bucket argument | n/a | raises `ValueError` | + +**Importing it emits a `DeprecationWarning`.** Migrate to +`FilesManager("patient")` or, better, +`django.core.files.storage.storages["patient"]`. + +**The signed-URL removal is deliberate and will not be restored.** ADR-0001 +requires that no storage-provider URL reach a client. A plugin that needs to +hand a file to a browser should link to the CARE download route rather than mint +a bucket URL: + +```http +GET /api/v1/files/{external_id}/download/ +``` + +**verified** Nothing prevents a plugin from importing `boto3` itself and +generating its own presigned URL — `boto3` remains a core dependency for SNS. +The guarantee ES-01 establishes is that *CARE* generates none; it is not +enforced against plugin code. **Recommend** adding this to plugin review +criteria rather than attempting to block the import. + +**unknown** Which plugins, if any, import `S3FilesManager`. No plugin source is +vendored here, so the shim is retained on the assumption that some do. diff --git a/docs/xii/architecture/inventory/runtime-and-deployment.md b/docs/xii/architecture/inventory/runtime-and-deployment.md new file mode 100644 index 0000000000..1c6304c543 --- /dev/null +++ b/docs/xii/architecture/inventory/runtime-and-deployment.md @@ -0,0 +1,683 @@ +--- +title: Runtime and Deployment Inventory +document: inventory/runtime-and-deployment +version: 0.2.0 +status: Draft +phase: 0 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +baseline_commit: 2fe40cd16 +reviewed: 2026-08-06 +--- + +# Runtime and Deployment Inventory + +How CARE is built, started and tested today, plus the Phase 0 runtime baseline +(§11). Nothing in the runtime was modified in this phase. + +Evidence labels: **verified** / **inferred** / **unknown**. + +--- + +## 1. Process commands + +**verified** Five entrypoint scripts define every process CARE runs. + +| Process | Script | Command | Used by | +| --- | --- | --- | --- | +| API (prod) | `scripts/start.sh` | `gunicorn --config python:config.gunicorn config.wsgi:application --bind 0.0.0.0:9000 --chdir=/app --workers $GUNICORN_WORKERS` | `docker/prod.Dockerfile` | +| API (dev) | `scripts/start-dev.sh` | `python manage.py runserver_plus 0.0.0.0:9000 --print-sql` | `docker-compose.local.yaml:13` | +| Worker (prod) | `scripts/celery_worker.sh` | `celery --app=config.celery_app worker --max-tasks-per-child=6 --loglevel=info --concurrency=${CELERY_WORKER_CONCURRENCY:-1}` | — | +| Beat (prod) | `scripts/celery_beat.sh` | `celery --app=config.celery_app beat --loglevel=info` | — | +| Worker+Beat (dev) | `scripts/celery-dev.sh` | `watchmedo auto-restart ... celery ... worker -B --loglevel=INFO` | `docker-compose.local.yaml:30` | + +**verified** ECS variants also exist: `scripts/start-ecs.sh`, +`scripts/celery_worker-ecs.sh`, `scripts/celery_beat-ecs.sh`. + +**verified** `scripts/celery-dev.sh` runs `worker -B` — worker and beat in one +process. `scripts/celery_worker.sh` and `scripts/celery_beat.sh` separate them. + +**verified** `Procfile` describes a third shape entirely: + +```text +web: gunicorn config.wsgi:application +release: python manage.py collectstatic --noinput && python manage.py migrate +``` + +**verified** The `Procfile` `release` phase is the **only** place in the +repository where migrations are tied to a deploy step rather than to a +long-running process. It defines no worker. + +--- + +## 2. Migration behavior + +**verified** This is the single most important runtime fact for Cloud Run. + +| Script | Runs `migrate`? | Line | +| --- | --- | --- | +| `scripts/start.sh` (API, prod) | **no** | — | +| `scripts/start-dev.sh` (API, dev) | **no** | — | +| `scripts/celery_worker.sh` | **no** | — | +| `scripts/celery_beat.sh` | **yes** | `python manage.py migrate --noinput` | +| `scripts/celery-dev.sh` | **yes** | `python manage.py migrate --noinput` | +| `Procfile` | yes, as `release` | line 2 | + +**verified** In the Docker-based deployment, **schema migration is a side effect +of starting Celery Beat**. The API container never migrates. + +**verified** `scripts/celery_beat.sh` and `scripts/celery-dev.sh` also run two +data-seeding commands after migrating: + +```bash +python manage.py sync_permissions_roles +python manage.py sync_valueset +``` + +**verified** `care/security/management/commands/sync_permissions_roles.py:14` +documents that concurrent runs are *"automatically blocked with redis"* — i.e. +this startup step depends on the distributed lock described in +`cache-and-redis.md` §4.2. + +**inferred** Cloud Run has no beat process. Migrations and both sync commands +need an explicit home — a Cloud Run Job or a deploy step — or they will never +run. This is not a refactor; it is a gap that appears the moment beat is removed. + +--- + +## 3. Startup dependencies + +**verified** Every prod and dev entrypoint waits for **both** PostgreSQL and +Redis before starting: + +| Script | `wait_for_db.sh` | `wait_for_redis.sh` | +| --- | --- | --- | +| `scripts/start.sh` | yes | **yes** | +| `scripts/start-dev.sh` | yes | **yes** | +| `scripts/celery_worker.sh` | yes | yes | +| `scripts/celery_beat.sh` | yes | yes | +| `scripts/celery-dev.sh` | yes | yes | + +**verified** The API blocks on Redis at startup even though its only Redis use is +the cache and the token denylist. + +**inferred** On Cloud Run this is a cold-start blocker: an instance cannot serve +until Redis answers. Combined with `IGNORE_EXCEPTIONS: True` +(`config/settings/base.py:93`), the runtime is inconsistent — it refuses to +*start* without Redis but silently tolerates Redis failing later. + +**verified** `scripts/start.sh`, `celery_worker.sh` and `celery_beat.sh` all +synthesize connection URLs when unset: + +```bash +export DATABASE_URL="postgres://${POSTGRES_USER}:${POSTGRES_PASSWORD}@${POSTGRES_HOST}:${POSTGRES_PORT}/${POSTGRES_DB}" +export REDIS_URL="rediss://:${REDIS_AUTH_TOKEN}@${REDIS_HOST}:${REDIS_PORT}/${REDIS_DATABASE}?ssl_cert_reqs=none" +``` + +**verified** The Redis URL uses the TLS scheme `rediss://` with +`ssl_cert_reqs=none` — TLS without certificate verification. + +**Recommend** treating this as a defect to fix, not a baseline to carry forward. +`ssl_cert_reqs=none` encrypts the connection but authenticates nothing, so it +stops passive sniffing and not an active man-in-the-middle — which is most of +what TLS to a managed Redis endpoint is for. The target runtime SHALL verify +against the deployment CA (`ssl_cert_reqs=required` plus `ssl_ca_certs`, or the +system trust store where the provider uses a public CA). If verification must be +disabled anywhere, it SHALL be scoped to local development, where the endpoint +is a container on a private network and there is no CA to verify against. + +--- + +## 4. Static files and i18n + +**verified** `collectstatic --noinput` runs **at container start**, not at image +build time: + +| Script | Line | +| --- | --- | +| `scripts/start.sh` | `python manage.py collectstatic --noinput` | +| `scripts/start-dev.sh` | same | +| `scripts/celery_worker.sh` | same | + +**verified** `compilemessages -v 0` runs at start in all five scripts. + +**verified** `whitenoise` is a dependency (`Pipfile`, `whitenoise = "==6.11.0"`), +so static files are served from the application process. + +**inferred** Running `collectstatic` and `compilemessages` on every cold start +adds latency to every Cloud Run instance launch and repeats identical work. +Moving both into the image build is a contained, low-risk improvement. + +**verified** `scripts/celery_worker.sh` runs `collectstatic` even though a worker +serves no HTTP. + +--- + +## 5. Health checks + +**verified** `docker/prod.Dockerfile:65-70` declares: + +```dockerfile +HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=12 CMD ["./healthcheck.sh"] +``` + +**verified** `scripts/healthcheck.sh` dispatches on a role file written by each +entrypoint to `/tmp/container-role`: + +| Role | Probe | +| --- | --- | +| `api` | `curl -fsS http://localhost:9000/ping/` | +| `celery-beat` | `ls /tmp/healthy` — a marker file touched before beat starts | +| `celery*` | `celery -A config.celery_app inspect ping -d celery@$HOSTNAME` | + +**verified** Roles are written at the top of each script: `api` +(`start.sh`, `start-dev.sh`), `celery` (`celery-dev.sh`), `celery-worker` +(`celery_worker.sh`), `celery-beat` (`celery_beat.sh`). + +**verified** The beat health check is a **liveness lie**: `touch /tmp/healthy` +happens *before* `celery beat` is exec'd in `scripts/celery_beat.sh`, so the file +persists even if beat dies. + +**verified** The application-level health config is `HEALTHY_DJANGO` at +`config/settings/base.py:453-467`, with three probes: database, cache, and Celery +queue length. See `cache-and-redis.md` §4.9 — the third connects directly to +Redis and is meaningless under Cloud Tasks. + +**inferred** For Cloud Run, only the `api` branch is relevant; `/ping/` is the +natural startup and liveness probe. + +--- + +## 6. Images + +**verified** Two Dockerfiles: `docker/dev.Dockerfile` and `docker/prod.Dockerfile`. + +**verified** `docker/prod.Dockerfile` structure: + +| Stage | Lines | Purpose | +| --- | --- | --- | +| `base` | 1-15 | `python:3.13-slim-bookworm`, env setup | +| `builder` | 19-39 | build deps, `pipenv install --deploy --categories "packages"`, plugin install | +| `runtime` | 42-72 | runtime deps, non-root `django` user, venv copy, healthcheck | + +**verified** Notable facts: + +- Base image `python:3.13-slim-bookworm` (`:1`) — matches `Pipfile`'s + `python_version = "3.13"`. +- Runs as non-root `django` (`:44-45`, `:63`). +- `EXPOSE 9000` (`:72`). +- **No `CMD` or `ENTRYPOINT`.** The image declares neither; the orchestrator must + supply the command. **inferred** Cloud Run requires an explicit container + command, so each service must set it. +- WeasyPrint native deps (`libpango`, `libharfbuzz`) installed in both stages + (`:23`, `:48`) — these are what make report generation work. +- Plugins are installed **at image build time** (`:34-39`) via + `install_plugins.py`, parameterized by the `ADDITIONAL_PLUGS` build arg + (`:37-38`). + +**verified** The `CMD` entries at `docker/prod.Dockerfile:70` and +`docker/dev.Dockerfile:36` are the **`HEALTHCHECK` `CMD`**, not a container +command. Neither image declares a top-level `CMD` or `ENTRYPOINT`. + +### 6.1 Production image availability + +**verified** `.github/workflows/deploy.yml` publishes a production image: + +| Fact | Evidence | +| --- | --- | +| Registry | `ghcr.io/${{ github.repository }}` → `ghcr.io/ohcnetwork/care` (`deploy.yml:64, 97`) | +| Dockerfile | `docker/prod.Dockerfile` (`deploy.yml:94`) | +| Architectures | `linux/amd64` + `linux/arm64` (`deploy.yml:44-49`) | +| Push mode | `push-by-digest=true,name-canonical=true,push=true` (`deploy.yml:97`) | +| Triggers | tags `v*`, pushes to `develop`, manual dispatch (`deploy.yml:3-11`) | +| Gate | `github.repository == 'ohcnetwork/care'` (`deploy.yml:31`) | + +**inferred** A ready-made multi-arch production image exists upstream, so a GCP +deployment can consume `ghcr.io/ohcnetwork/care` directly rather than building +one — provided it supplies its own container command (§6) and its own migration +step (§2). Note the fork's images are **not** published: the gate at +`deploy.yml:31` restricts the job to the upstream repository. + +**verified** The ECS deployment env block in `deploy.yml:19-29` is **entirely +commented out**, as is a further block at `deploy.yml:207-209`. **inferred** the +ECS deploy path is currently inactive upstream. + +--- + +## 7. Compose topology + +**verified** `docker-compose.yaml` defines three infrastructure services: + +| Service | Image | Host port | Healthcheck | +| --- | --- | --- | --- | +| `db` | `postgres:17-alpine` | 5433→5432 | `pg_isready` | +| `redis` | `redis:8-alpine` | 6380→6379 | `redis-cli ping` | +| `minio` | `minio/minio:latest` | 9100→9000, 9001 | `/minio/health/ready` | + +**verified** `docker-compose.local.yaml` adds `backend` and `celery`, both from +the `care_local` image built from `docker/dev.Dockerfile`. + +**verified** `backend` depends on `celery` with `condition: service_healthy` +(`docker-compose.local.yaml:23-24`). **inferred** this ordering exists because +`celery-dev.sh` runs the migrations — the API waits for the schema. + +**verified** MinIO buckets are created by `docker/minio/init-script.sh`, which +also sets them **public**: `mc anonymous set public local/$BUCKET_NAME` +(`init-script.sh:47`). + +**verified** PostgreSQL in compose is **17**; the target described in the +architecture docs is Cloud SQL. Version parity is a deployment decision, not a +code constraint. + +**verified** Additional compose files: `docker-compose.pre-built.yaml` and +`docker-compose.coolify.yaml`. + +--- + +## 8. CI + +**verified** `.github/workflows/` contains 8 workflows: `deploy.yml`, `docs.yml`, +`linter.yml`, `release.yml`, `reusable-test.yml`, `test-merge-queue.yml`, +`test-pull-request.yml`, `validate-pr-title.yml`. + +**verified** `reusable-test.yml` is the test pipeline. Its ordered steps: + +| Step | Command | Line | +| --- | --- | --- | +| Build image | `docker buildx build --file docker/dev.Dockerfile --tag care_local ... --platform linux/arm64` | 52-59 | +| Start services | `docker compose -f docker-compose.yaml -f docker-compose.local.yaml up -d --wait` | 63 | +| Check migrations | `make checkmigration` | 67 | +| Fixtures | `make load-fixtures` | 70 | +| Tests | `make test-coverage` | 77 | + +**verified** Runner is `ubuntu-24.04-arm` (`:17`); CI builds and tests +**arm64 only** (`:57`). + +**verified** `make checkmigration` → `python manage.py makemigrations --check --dry-run` +(`Makefile:47-48`). CI fails on uncommitted model changes. + +**verified** `make test-coverage` → `coverage run manage.py test --settings=config.settings.test --keepdb --parallel --shuffle` (`Makefile:63-66`). + +**verified** `config/settings/test.py:45-46` points the cache at +`django_redis.cache.RedisCache` on `REDIS_URL`, so **CI requires a live Redis**. + +--- + +## 9. Dependencies + +**verified** Manager: **Pipenv** (`Pipfile` + `Pipfile.lock`). No +`requirements.txt`, no Poetry, no uv. + +**verified** Key pins: + +| Package | Version | Relevance | +| --- | --- | --- | +| `python_version` | `3.13` | `Pipfile [requires]` | +| `django` | `==6.0` | | +| `celery` | `==5.6.0` | | +| `django-redis` | `==6.0.0` | supplies `delete_pattern`, `nx=`, `get_redis_connection` | +| `redis` | `==7.1.0` (extras `hiredis`) | | +| `boto3` | `==1.43.6` | only S3 client today | +| `psycopg` | `==3.3.2` (extras `c`) | | +| `gunicorn` | `==23.0.0` | | +| `whitenoise` | `==6.11.0` | | +| `django-ratelimit` | `==4.1.0` | | +| `healthy-django` | `==0.1.0` | | +| `weasyprint` | `==68.0` | report rendering | +| `drf-spectacular` | `==0.29.0` | | +| `sentry-sdk` | `==2.58.0` | | + +**verified** `pyproject.toml:22` sets `requires-python = "==3.13.*"` and +`pyproject.toml:58` sets ruff `target-version = "py313"`. + +**verified** **Absent** from `Pipfile`: `django-storages`, any +`google-cloud-*` package, `django-celery-beat`, `django-celery-results`. + +**Superseded by IS-01 for the first two.** `Pipfile:55` now carries +`django-storages = {extras = ["s3", "google"], version = "==1.14.6"}`, which +brings `google-cloud-storage` in transitively. `django-celery-beat` and +`django-celery-results` remain absent. + +**verified** `django-anymail` is installed with the `amazon-ses` extra. +**inferred** email delivery is AWS SES today; GCP has no drop-in equivalent, so +this needs an explicit decision. + +--- + +## 10. Existing GCP-related code + +**verified** A repository-wide search for `gcp`, `google.cloud`, `cloud run`, +`cloudsql`, `cloud_tasks`, `django-storages` and `django_storages` across +`*.py`, `*.yml`, `*.yaml`, `*.sh`, `Pipfile` and `*.toml`, excluding +`docs/xii/`, returns exactly **two** matches: + +| File | Line | Content | +| --- | --- | --- | +| `care/utils/csp/config.py` | 20 | `GCP = "GCP"` — a `CSProvider` enum member | +| `care/emr/utils/file_manager.py` | 130 | `# bulk delete is not supported by some providers: GCP` | + +**verified** The `CSProvider.GCP` member is **never branched on**. +`BUCKET_PROVIDER` is only ever compared against `CSProvider.AWS_ROLE_BASED` +(`care/utils/csp/config.py:35, 48, 62`). + +**verified** There is no Terraform, no Cloud Build config, no `app.yaml`, no +`service.yaml`, and no GCP credentials handling anywhere in the repository. + +**Conclusion (verified):** GCP support does not exist. This is genuinely +greenfield. + +**Superseded by IS-01.** Both rows above are gone: `care/utils/csp/config.py` was +deleted with the provider-specific bucket configuration, and the `file_manager.py` +bulk-delete comment went with the boto3 code. Re-running the same search now +returns matches in `config/storage.py`, `config/settings/base.py`, +`care/utils/tests/test_storage_config.py` and `Pipfile` — the `gcs` backend +option and its tests. + +The conclusion still holds for everything outside storage: there is still no +Terraform, no Cloud Build config, no `app.yaml`, no `service.yaml` and no GCP +credentials handling. Storage is the one axis where GCP is now selectable, and +selecting it is a settings change rather than a code change. + +--- + +## 11. Baseline command results + +**Status: GREEN.** Recorded 2026-08-06. The blockers listed in the previous +revision of this section are resolved; the record below supersedes them. + +No destructive command was run: no volume was removed, no database was reset, no +container or image existed before the run. `docker volume ls`, `docker ps -a` and +`docker images` were all empty at the start, so every artifact below was created +by this baseline. + +### 11.1 Environment + +| Field | Value | +| --- | --- | +| Operating system | Microsoft Windows 11 Pro, version 10.0.26200 | +| Shell | PowerShell 5.1.26100.8875; Git Bash for the `docker compose` invocations | +| Docker Engine | client **29.6.2**, server **29.6.2** (Docker Desktop, WSL2 backend) | +| Docker Compose | **v5.3.1** | +| Container kernel | `Linux-6.6.114.1-microsoft-standard-WSL2-x86_64-with-glibc2.36` | +| Repository branch | `feature/gcp-phase-0-inventory` | +| Repository commit | `2fe40cd16` | +| Working tree | clean before and after | +| Python in image | 3.13.14 | +| Django in image | 6.0 | +| Plugins | none — `plug_config.py` declares `plugs = []`, `ADDITIONAL_PLUGS` unset | + +**verified** Docker Desktop upgraded its own components when it was launched: the +CLI reported `29.5.2` / Compose `v5.1.4` before the daemon started and +`29.6.2` / `v5.3.1` afterwards. The versions in the table are the ones the +baseline actually ran on. + +### 11.2 Environment files + +**verified** No gitignored environment file had to be created. This corrects the +previous revision, which listed a missing `.env` as a blocker. + +| File | Status | Role | +| --- | --- | --- | +| `docker/.local.env` | **tracked in git** | `env_file` for `backend` and `celery` (`docker-compose.local.yaml:10, 29`) | +| `docker/.prebuilt.env` | **tracked in git** | `env_file` for `db` (`docker-compose.yaml:11`) | +| `.env` (repository root) | gitignored, **absent, not required** | — | + +**verified** There is no `docker/.local.env.example` and no +`docker/.prebuilt.env.example`. The two `.env` files are the real, committed +artifacts, not templates. + +**verified** `docker compose config` resolves with **no** missing-variable +warnings without a root `.env`. Every interpolation in the compose files supplies +a default: `BACKUP_DIR` (`docker-compose.yaml:14`), `MINIO_ACCESS_KEY` and +`MINIO_SECRET_KEY` (`:42-43`), `POSTGRES_USER` (`:18`), and `ADDITIONAL_PLUGS` +(`docker-compose.local.yaml:8`). + +**verified** Compose v5.3.1 interpolates values *inside* `env_file`. The literal +`BUCKET_KEY=${MINIO_ACCESS_KEY:-minioadmin}` at `docker/.local.env:14` arrives in +the container as `BUCKET_KEY=minioadmin`. Confirmed by reading the resolved +environment inside `backend`. + +### 11.3 Commands executed + +**verified** GNU Make is not installed on this host, so each `Makefile` target +was translated to the exact `docker compose` command it wraps. File order and +flags are unchanged from the `Makefile`. + +| Step | `Makefile` target | Command executed | +| --- | --- | --- | +| Build | `build` (`:19-20`) | `docker compose -f docker-compose.yaml -f docker-compose.local.yaml build` | +| Start + wait | `up` (`:25-26`) | `docker compose -f docker-compose.yaml -f docker-compose.local.yaml up -d --wait` | +| Service state | `list` (`:41-42`) | `docker compose -f docker-compose.yaml -f docker-compose.local.yaml ps` | +| Migration check | `checkmigration` (`:47-48`) | `docker compose exec backend bash -c "python manage.py makemigrations --check --dry-run"` | +| Fixtures | `load-fixtures` (`:38-39`) | `docker compose exec backend bash -c "python manage.py load_fixtures"` | +| Tests | `test` (`:56-57`) | `docker compose exec backend bash -c "python manage.py test --keepdb --parallel --shuffle"` | + +**verified** The four `exec` targets in the `Makefile` pass **no** `-f` flags, so +they rely on Compose's default file resolution — which finds only +`docker-compose.yaml`, where `backend` is not defined. This works anyway: +Compose v5 `exec` resolves the container by project label (project name `care`, +derived from the directory), not by service presence in the loaded config. +Confirmed empirically — `docker compose exec backend bash -c "echo OK"` succeeds. +**inferred** This is an implicit dependency on Compose's lookup behaviour rather +than an intentional design, but it is not currently broken. + +### 11.4 Build result + +**verified** `care_local:latest` built successfully. + +| Field | Value | +| --- | --- | +| Image | `care_local:latest` | +| Manifest list digest | `sha256:4ea640b476e9050288e7f98899d848a987339e56e78ff3c4b75b3a96a8b6f70b` | +| Size | 1.6 GB | +| Platform | `linux/amd64` | +| Dockerfile | `docker/dev.Dockerfile` | + +**The first build attempt failed.** Classified as a **dependency-build** failure, +not an application defect: + +```text +zipfile.BadZipFile: Bad CRC-32 for file '_brotli.cpython-313-x86_64-linux-gnu.so' +ERROR: Couldn't install package: {} +failed to solve: process "/bin/sh -c pipenv install --system --categories \"packages dev-packages docs\"" + did not complete successfully: exit code: 1 +``` + +A corrupted wheel had been written into the BuildKit pip cache mount declared at +`docker/dev.Dockerfile:22`. Because the cache mount persists across builds, a +plain retry would have reused the same corrupt file. The minimum correction was +to drop only the cache mounts — +`docker builder prune --filter type=exec.cachemount` (92.29 MB, all of it created +minutes earlier by that same failed build). No volume, container or image was +touched. The rebuild succeeded and the failure has not recurred. **inferred** +transient; no source change was made or needed. + +### 11.5 Service health + +**verified** All five services reached `healthy` under `up -d --wait`, and were +still healthy an hour later. + +| Service | Container | Health | Ports | +| --- | --- | --- | --- | +| `db` | `care-db-1` | healthy | 5433→5432 | +| `redis` | `care-redis-1` | healthy | 6380→6379 | +| `minio` | `care-minio-1` | healthy | 9100→9000, 9001→9001 | +| `celery` | `care-celery-1` | healthy | — | +| `backend` | `care-backend-1` | healthy | 9000→9000, 9876→9876 | + +**verified** `curl http://localhost:9000/ping/` inside `backend` returns +`{"status": "OK"}`. + +### 11.6 Startup sequence + +**verified** from container logs, in order. + +`celery` (`scripts/celery-dev.sh`): + +| Step | Evidence | +| --- | --- | +| Waited for PostgreSQL | `Waiting for PostgreSQL to become available...` ×2, then `PostgreSQL is available` | +| Waited for Redis | `Redis is available` | +| Ran migrations | `Running migrations:` — **312** `Applying ... OK` lines across `admin, auth, authtoken, contenttypes, emr, facility, security, sessions, sites, users` | +| Ran `sync_permissions_roles` | no stdout; verified by effect (§11.7) | +| Ran `sync_valueset` | no stdout; verified by effect (§11.7) | +| Worker started | banner `celery@b2c90b2c6f0b v5.6.0`, `concurrency: 16 (prefork)`, `transport: redis://redis:6379/0`, 8 registered tasks, `Connected to redis://redis:6379/0` | +| Beat started | `worker -B`; `/app/celerybeat-schedule`, `-shm` and `-wal` present and being written | + +`backend` (`scripts/start-dev.sh`): + +| Step | Evidence | +| --- | --- | +| Waited for PostgreSQL | `PostgreSQL is available` | +| Waited for Redis | `Redis is available` | +| `collectstatic` | `198 static files copied to '/app/staticfiles', 926 post-processed.` | +| Server started | `starting server...`, health check passing on `/ping/` | + +**verified** The database did not exist beforehand; `scripts/wait_for_db.sh` +created it (`Creating Database` path) before migrations ran. + +**verified — minor logging gap.** The `celery` log ends at +`mingle: all alone` and never emits the usual `celery@ ready.` line, nor +any `beat: Starting...` line. The worker is nonetheless live — +`celery -A config.celery_app inspect ping` returns `1 node online` — and beat is +live, evidenced by the schedule files above. **inferred** log truncation under +`watchmedo auto-restart`, not a process failure. Recorded so that a future reader +does not mistake the missing lines for a broken worker. + +### 11.7 Migration and synchronization results + +| Check | Result | +| --- | --- | +| `migrate` | **312** migrations applied, 0 errors | +| `makemigrations --check --dry-run` | `No changes detected` — no model drift | +| `sync_permissions_roles` | `security_permissionmodel` = **115**, `security_rolemodel` = **10**, `security_rolepermission` = **546** | +| `sync_valueset` | `emr_valueset` = **30** | + +**verified** Neither sync command prints to stdout. Both run under +`set -euo pipefail` in `scripts/celery-dev.sh`, so a failure would have aborted +container startup; their success is additionally confirmed by the row counts +above, queried directly from `care-db-1`. + +### 11.8 Fixture result + +**verified** `python manage.py load_fixtures` completed successfully: +`All fixtures loaded successfully!` + +Seventeen fixture groups loaded — organizations, facility, departments, +locations, devices, users, patients, encounters, facility organization +memberships, secondary facility, questionnaires, report templates, lab +definitions, inventory, billing, scheduling, managing organization. + +Resulting counts: `users_user` = 10, `facility_facility` = 2, +`emr_organization` = 12. Ten test accounts are printed by the command; the +credentials are development-only and are not reproduced here. + +### 11.9 Test results + +**verified** Command: +`docker compose exec backend bash -c "python manage.py test --keepdb --parallel --shuffle"` + +Settings module is `config.settings.test`, selected automatically by +`manage.py:15-16`. `--parallel` used **16** workers (17 test databases including +the primary). + +| Run | Shuffle seed | Tests | Pass | Fail | Skip | Test duration | Wall clock | Exit | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 1 | 8926493199 | 1912 | 1911 | 1 | 0 | 22.296 s | 138 s | 1 | +| 2 | 8078922123 | 1912 | 1912 | 0 | 0 | 29.482 s | 48 s | 0 | +| 3 | 3975608265 | 1912 | 1912 | 0 | 0 | 21.158 s | 37 s | 0 | + +**Test count: 1912. Skipped: 0. Expected failures: 0. Warnings: none emitted by +the test runner.** + +Run 1's longer wall clock includes first-time creation of the 17 test databases; +runs 2 and 3 reused them via `--keepdb`. + +**Baseline verdict: green.** Two of three runs are fully clean; the single +failure in run 1 is a test-isolation flake in the suite itself, characterised +below, not a defect in the application under test. + +### 11.10 Known defect — flaky rate-limit test + +**verified** Run 1 failed one test: + +```text +care/emr/tests/test_reset_password_api.py:375 + ResetPasswordAPITest.test_password_request_rate_limiting +AssertionError: 200 != 429 +``` + +**verified** It passes deterministically in isolation, both as a single test and +as the whole module, run serially: +`python manage.py test care.emr.tests.test_reset_password_api --keepdb` → `Ran 23 tests ... OK`. + +**verified** Mechanism — two facts combine: + +1. `config/ratelimit.py:9`, `get_ratelimit_key`, returns the **constant** string + `"ratelimit"`. The rate-limit counter is therefore process-global: it is not + keyed by IP, user or test. +2. `config/settings/test.py:45-56` points the cache at **Redis**, shared by all + 16 parallel workers under a single `KEY_PREFIX` of `test_`. Meanwhile + `cache.clear()` runs in `setUp` at `care/emr/tests/test_reset_password_api.py:24` + and `care/emr/tests/test_valueset_api.py:23, 52`. + +**inferred** A concurrent worker calling `cache.clear()` wipes the shared global +counter partway through the test's 11-request loop, so the final request returns +200 instead of 429. `--shuffle` decides whether the interleaving happens, which +is why the failure is intermittent. + +**Not fixed.** This is a pre-existing upstream test-isolation defect, unrelated to +GCP work, and fixing it would be an application change outside the scope of +recording a baseline. It is reproducible in principle on upstream CI, which runs +the same `--parallel --shuffle` combination via `make test-coverage` +(`.github/workflows/reusable-test.yml:77`). + +**inferred, relevant to the target runtime:** the global rate-limit key is also a +correctness concern beyond tests — it means the limit is shared across all +callers, not per client. Recorded here as an observation only. + +### 11.11 Reproducing this baseline + +```bash +# 1. build +docker compose -f docker-compose.yaml -f docker-compose.local.yaml build + +# 2. start and wait for health +docker compose -f docker-compose.yaml -f docker-compose.local.yaml up -d --wait + +# 3. confirm state +docker compose -f docker-compose.yaml -f docker-compose.local.yaml ps + +# 4. migrations already ran in the celery container; confirm no drift +docker compose exec backend bash -c "python manage.py makemigrations --check --dry-run" + +# 5. fixtures +docker compose exec backend bash -c "python manage.py load_fixtures" + +# 6. tests +docker compose exec backend bash -c "python manage.py test --keepdb --parallel --shuffle" +``` + +No `.env` file is needed. No teardown, volume deletion or database reset is +required or advised — `make teardown` (`Makefile:34-35`) and `make reset-db` +(`:76-78`) are destructive and were deliberately not used. + +--- + +## 12. Runtime facts most relevant to Cloud Run + +**verified**, ordered by how much they constrain the design: + +1. **Migrations run only in Celery Beat startup** (§2). Removing beat removes + migrations. +2. **`sync_permissions_roles` and `sync_valueset` run at beat startup** (§2) and + the first depends on a Redis lock. +3. **The API blocks on Redis before serving** (§3). +4. **The prod image declares no `CMD`** (§6). Every Cloud Run service must set one. +5. **`collectstatic` and `compilemessages` run per cold start** (§4). +6. **Celery timezone is hardcoded to `Asia/Kolkata`** (`config/celery_app.py:16`); + any Cloud Scheduler translation must account for IST. +7. **The Celery queue-length health check binds to Redis** (§5). +8. **CI builds arm64 only** (§8); Cloud Run defaults to amd64. +9. **`django-anymail[amazon-ses]`** (§9) ties email to SES. diff --git a/docs/xii/architecture/inventory/storage-call-sites.md b/docs/xii/architecture/inventory/storage-call-sites.md new file mode 100644 index 0000000000..f5a7643117 --- /dev/null +++ b/docs/xii/architecture/inventory/storage-call-sites.md @@ -0,0 +1,538 @@ +--- +title: Storage Call-Site Inventory +document: inventory/storage-call-sites +version: 0.2.0 +status: Draft +phase: 1 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +reviewed: 2026-08-06 +--- + +# Storage Call-Site Inventory + +Every object-storage call site in the repository, enumerated. + +Sections 1-8 are the Phase 0 snapshot, recorded before any storage code was +modified. **Section 11 records what IS-01 changed** and marks the migration +status of every call site. Where the two disagree, section 11 is current. + +Evidence labels used throughout: + +- **verified** — read directly from the file at the stated line. +- **inferred** — deduced from surrounding code; not directly asserted anywhere. +- **unknown** — cannot be determined from this repository alone. + +--- + +## 1. Summary + +**verified** The entire object-storage surface is 4 source files: + +| File | Role | +| --- | --- | +| `care/emr/utils/file_manager.py` | The only S3 abstraction (`S3FilesManager`) | +| `care/utils/csp/config.py` | Credential/endpoint/bucket resolution | +| `care/utils/file_uploads/cover_image.py` | Direct `boto3` use, bypasses `S3FilesManager` | +| `config/settings/base.py` | Bucket settings | + +**verified** Total distinct storage call sites: **19** (counted in §3). + +**verified** `boto3` is imported in exactly 3 non-test modules: +`care/emr/utils/file_manager.py:3`, `care/utils/file_uploads/cover_image.py:5`, +and `care/utils/sms/backend/sns.py:8` (SNS, not storage — out of scope). + +--- + +## 2. Bucket topology + +**verified** `care/utils/csp/config.py:27-30` defines three logical bucket types: + +```text +BucketType.PATIENT +BucketType.FACILITY +BucketType.REPORT +``` + +**verified** These map to only **two** physical buckets +(`care/utils/csp/config.py:33-70`): + +| BucketType | Physical bucket setting | Resolver | Line | +| --- | --- | --- | --- | +| `FACILITY` | `settings.FACILITY_S3_BUCKET` | `get_facility_bucket_config` | `config.py:43` | +| `PATIENT` | `settings.FILE_UPLOAD_BUCKET` | `get_patient_bucket_config` | `config.py:56` | +| `REPORT` | `settings.FILE_UPLOAD_BUCKET` | `get_report_bucket_config` | `config.py:70` | + +**verified** `REPORT` and `PATIENT` resolve to the *same* physical bucket. This is +intentional and documented in the docstring at `config.py:60`: +`"""Get bucket configuration for reports - uses same bucket as patient files"""`. + +### 2.1 Defect: patient/report buckets use facility credentials + +**verified** `get_patient_bucket_config` (`config.py:46-56`) and +`get_report_bucket_config` (`config.py:59-70`) read +`FACILITY_S3_REGION`, `FACILITY_S3_KEY` and `FACILITY_S3_SECRET` — +not the `FILE_UPLOAD_*` equivalents: + +```python +# care/utils/csp/config.py:46-56 +def get_patient_bucket_config(external) -> tuple[ClientConfig, BucketName]: + params = {"region_name": settings.FACILITY_S3_REGION} # line 47 + if CSProvider.AWS_ROLE_BASED.value != settings.BUCKET_PROVIDER: + params["aws_access_key_id"] = settings.FACILITY_S3_KEY # line 49 + params["aws_secret_access_key"] = settings.FACILITY_S3_SECRET # line 50 + ... + return params, settings.FILE_UPLOAD_BUCKET # line 56 +``` + +**verified** Consequence: `FILE_UPLOAD_REGION` (`base.py:537`), +`FILE_UPLOAD_KEY` (`base.py:538`) and `FILE_UPLOAD_SECRET` (`base.py:539`) are +defined but **read by no code in this repository**. A repository-wide grep for +each of those three names returns only their definition lines. + +**verified** The endpoint settings are *not* affected — `config.py:52-55` and +`config.py:66-69` correctly use `FILE_UPLOAD_BUCKET_ENDPOINT` / +`FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT`. + +**Migration consequence (inferred):** separate credentials per bucket cannot be +configured today. Any GCP design that assumes distinct service accounts or HMAC +keys per bucket must fix `config.py` first, or accept a single shared credential. +Recorded in `unresolved-items.md` §2. + +--- + +## 3. Call-site table + +Surface legend: `API` = request/response path, `TASK` = Celery task, +`CMD` = management command, `PLUGIN` = plugin-owned. + +| # | File | Line | Symbol | Operation | Bucket | Surface | Exposes direct object-storage URL | Full read into memory | Replaceable by Django Storage API | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 1 | `care/emr/utils/file_manager.py` | 35-50 | `S3FilesManager.signed_url` | `generate_presigned_url("put_object")` | per instance | API | **yes** | no | **no** — see §4.1 | +| 2 | `care/emr/utils/file_manager.py` | 52-69 | `S3FilesManager.read_signed_url` | `generate_presigned_url("get_object")` | per instance | API | **yes** | no | **no** — see §4.1 | +| 3 | `care/emr/utils/file_manager.py` | 71-79 | `S3FilesManager.put_object` | `put_object` | per instance | API + TASK | no | caller-dependent | **yes** — `Storage.save()` | +| 4 | `care/emr/utils/file_manager.py` | 81-88 | `S3FilesManager.get_object` | `get_object` | per instance | TASK + test | no | no (returns stream) | **yes** — `Storage.open()` | +| 5 | `care/emr/utils/file_manager.py` | 90-94 | `S3FilesManager.file_contents` | `get_object` + `Body.read()` | per instance | inferred: none | no | **yes** — `.read()` line 93 | **yes** — `Storage.open().read()` | +| 6 | `care/emr/utils/file_manager.py` | 96-110 | `S3FilesManager.delete_object` | `delete_object` | per instance | TASK | no | no | **yes** — `Storage.delete()` | +| 7 | `care/emr/utils/file_manager.py` | 112-133 | `S3FilesManager.delete_objects` | `delete_objects` (batch) | per instance | inferred: none | no | no | **partial** — see §4.2 | +| 8 | `care/utils/file_uploads/cover_image.py` | 14-21 | `delete_cover_image` | `delete_object` | `FACILITY` | API | no | no | **yes** | +| 9 | `care/utils/file_uploads/cover_image.py` | 24-53 | `upload_cover_image` | `delete_object` (line 35) + `put_object` (line 51) | `FACILITY` | API | no | no | **partial** — ACL, see §4.3 | +| 10 | `care/emr/models/file_upload.py` | 33 | `FileUpload.files_manager` | class attribute binding | `PATIENT` | — | — | — | n/a | +| 11 | `care/emr/models/report/report_upload.py` | 34 | `ReportUpload.files_manager` | class attribute binding | `REPORT` | — | — | — | n/a | +| 12 | `care/emr/api/viewsets/file_upload.py` | 262 | `FileUploadViewSet.upload_file` | `put_object` | `PATIENT` | API | no | **yes** — see §5.2 | **yes** | +| 13 | `care/emr/resources/file_upload/spec.py` | 115 | `FileUploadRetrieveSpec.perform_extra_serialization` | `signed_url` | `PATIENT` | API | **yes** | no | **no** | +| 14 | `care/emr/resources/file_upload/spec.py` | 117 | `FileUploadRetrieveSpec.perform_extra_serialization` | `read_signed_url` | `PATIENT` | API | **yes** | no | **no** | +| 15 | `care/emr/resources/report/report_upload/spec.py` | 52 | `ReportUploadRetrieveSpec.perform_extra_serialization` | `signed_url` | `REPORT` | API | **yes** | no | **no** | +| 16 | `care/emr/resources/report/report_upload/spec.py` | 54 | `ReportUploadRetrieveSpec.perform_extra_serialization` | `read_signed_url` | `REPORT` | API | **yes** | no | **no** | +| 17 | `care/emr/reports/report_utils.py` | 124-126 | `generate_and_upload_report` | `put_object` | `REPORT` | TASK | no | **yes** — `output_bytes` | **yes** | +| 18 | `care/emr/tasks/cleanup_incomplete_file_uploads.py` | 33 | `cleanup_incomplete_file_uploads` | `delete_object` | `PATIENT` | TASK | no | no | **yes** | +| 19 | `care/emr/reports/context_builder/data_points/fileupload.py` | 29 | data-point `mapping` lambda | `read_signed_url` | `PATIENT` | TASK | **yes** | no | **no** | + +### 3.1 Test-only call sites (not counted above) + +**verified** — listed for completeness, excluded from the 19: + +| File | Line | Symbol | +| --- | --- | --- | +| `care/emr/tests/test_file_upload_api.py` | 19 | `@override_settings(FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT=settings.BUCKET_ENDPOINT)` | +| `care/emr/tests/test_file_upload_api.py` | 77, 102, 137, 165 | asserts on `signed_url` / `read_signed_url` response fields | +| `care/emr/tests/test_file_upload_api.py` | 180 | `file_obj.files_manager.get_object(file_obj)` | + +--- + +## 4. Call sites that block a straight Django Storage swap + +### 4.1 Presigned URL generation (call sites 1, 2, 13-16, 19) + +**verified** `signed_url` and `read_signed_url` call +`boto3.client("s3").generate_presigned_url(...)` +(`file_manager.py:46` and `file_manager.py:61`). + +**verified** Django's `Storage` API has no presigned-URL method. `Storage.url()` +exists but its semantics are backend-defined, and `django-storages` implements +GCS signing through `GoogleCloudStorage.url()` rather than through a portable API. + +**verified** `signed_url` is a **write** URL — `generate_presigned_url("put_object", ...)` +at `file_manager.py:46-47`. This is the direct browser-to-bucket upload path that +the stated architecture goal excludes. + +**verified** `read_signed_url` sets `ResponseContentDisposition` +(`file_manager.py:66`) computed from a MIME allowlist `SAFE_INLINE_FORMATS` +(`file_manager.py:11-20`). Any Django-proxied download must reproduce this +`inline` vs `attachment` decision (`file_manager.py:57-59`) or it changes browser +behavior for patient documents. + +### 4.2 Batch delete (call site 7) + +**verified** `delete_objects` (`file_manager.py:112-133`) already contains a +GCP-specific branch: + +```python +# care/emr/utils/file_manager.py:128-133 +except ClientError as e: + if e.response["Error"]["Code"] == "NotImplemented": + # bulk delete is not supported by some providers: GCP + msg = f"Batch delete objects not implemented for {self.bucket_type.value} bucket" + raise NotImplementedError(msg) from e + raise +``` + +**verified** This comment at `file_manager.py:130` is one of only two pre-existing +GCP references in application code (the other is the `CSProvider.GCP` enum member +at `care/utils/csp/config.py:20`). + +**verified** No caller of `delete_objects` exists in this repository. A grep for +`delete_objects` returns only its definition (`file_manager.py:112`, `:123`) and +the prompt document. It is dead code as of this commit. + +### 4.3 ACL on cover-image upload (call site 9) + +**verified** `care/utils/file_uploads/cover_image.py:49-51`: + +```python +if settings.BUCKET_HAS_FINE_ACL: + boto_params["ACL"] = "public-read" +s3.put_object(**boto_params) +``` + +**verified** `BUCKET_HAS_FINE_ACL` is defined at `config/settings/base.py:531` +and read at exactly one place: `cover_image.py:49`. + +**inferred** GCS buckets with uniform bucket-level access reject per-object ACLs. +`django-storages` `GoogleCloudStorage` does not expose a per-object ACL parameter +in the same shape. This branch needs an explicit decision, not a mechanical port. + +--- + +## 5. Files read fully into memory + +**verified** Three call sites materialize whole file bodies: + +| Site | File | Line | What is buffered | +| --- | --- | --- | --- | +| 5 | `care/emr/utils/file_manager.py` | 93 | `response["Body"].read()` — entire object | +| 12 | `care/emr/api/viewsets/file_upload.py` | 224, 229 | base64-decoded request body | +| 17 | `care/emr/reports/report_utils.py` | 96, 124 | rendered report `output_bytes` | + +### 5.2 The `upload-file` endpoint already proxies through Django + +**verified** `FileUploadViewSet.upload_file` +(`care/emr/api/viewsets/file_upload.py:213-270`, route `POST /api/v1/files/upload-file/`) +accepts a **base64 string** in the JSON field `file_data` +(`file_upload.py:216`), decodes it (`file_upload.py:224`), wraps it in a +`ContentFile` (`file_upload.py:229`) and calls `put_object` +(`file_upload.py:262`). + +**verified** This means a Django-proxied upload path **already exists** and is +routed. It is not a greenfield addition. + +**verified** Constraints on that existing path: + +- `file_upload.py:224` — `base64.b64decode(file_data)` holds the whole file in + memory; base64 inflates the transferred body by ~33%. +- `file_upload.py:231-234` — size ceiling is `settings.MAX_FILE_UPLOAD_SIZE` MB, + defined at `config/settings/config.py:22` with a default of **5 MB**. +- `file_upload.py:237` — MIME sniffed from the first 2048 bytes via `magic.from_buffer`. +- `file_upload.py:242-244` — MIME checked against `settings.ALLOWED_MIME_TYPES`. +- `file_upload.py:255-268` — the DB row and the `put_object` share one + `transaction.atomic()` block. The storage write is **not** transactional; a + commit failure after a successful `put_object` orphans the object. See + `unresolved-items.md` §5. + +**inferred** For Cloud Run, the base64 body plus the in-memory decode sets the +per-request memory floor at roughly 2× file size. Cloud Run's 32 MiB default +request limit for HTTP/1 also caps effective upload size below +`MAX_FILE_UPLOAD_SIZE` unless HTTP/2 is used. Not verified against a deployed +service. + +--- + +## 6. Direct (unsigned) object-storage URLs + +**verified** Two call sites build bucket URLs by string concatenation, outside +`S3FilesManager` entirely: + +| File | Line | Symbol | +| --- | --- | --- | +| `care/facility/models/facility.py` | 211 | `Facility.read_cover_image_url` | +| `care/users/models.py` | 206 | `User.read_profile_picture_url` | + +```python +# care/facility/models/facility.py:207-212 +def read_cover_image_url(self): + if self.cover_image_url: + if settings.FACILITY_CDN: + return f"{settings.FACILITY_CDN}/{self.cover_image_url}" + return f"{settings.FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT}/{settings.FACILITY_S3_BUCKET}/{self.cover_image_url}" + return None +``` + +**verified** `care/users/models.py:202-207` is the same construction against the +same two settings. + +**verified** These URLs are unsigned and depend on the object being publicly +readable — which is what `BUCKET_HAS_FINE_ACL` / `ACL: public-read` +(`cover_image.py:49-51`) provides. + +**inferred** Facility cover images and user avatars are therefore *public* objects, +unlike patient files which are always signed. A GCP design that makes the bucket +uniformly private breaks both methods unless a Django-served route replaces them. + +--- + +## 7. Settings inventory + +**verified** All bucket settings, `config/settings/base.py`: + +| Setting | Line | Default | Read by | +| --- | --- | --- | --- | +| `BUCKET_PROVIDER` | 525 | `"aws"`, uppercased | `csp/config.py:35, 48, 62` | +| `BUCKET_REGION` | 526 | `"ap-south-1"` | defaults only | +| `BUCKET_KEY` | 527 | `""` | defaults only | +| `BUCKET_SECRET` | 528 | `""` | defaults only | +| `BUCKET_ENDPOINT` | 529 | `""` | defaults only | +| `BUCKET_EXTERNAL_ENDPOINT` | 530 | `BUCKET_ENDPOINT` | defaults only | +| `BUCKET_HAS_FINE_ACL` | 531 | `False` | `cover_image.py:49` | +| `FILE_UPLOAD_BUCKET` | 536 | `""` | `csp/config.py:56, 70` | +| `FILE_UPLOAD_REGION` | 537 | `BUCKET_REGION` | **nothing** | +| `FILE_UPLOAD_KEY` | 538 | `BUCKET_KEY` | **nothing** | +| `FILE_UPLOAD_SECRET` | 539 | `BUCKET_SECRET` | **nothing** | +| `FILE_UPLOAD_BUCKET_ENDPOINT` | 540-542 | `BUCKET_ENDPOINT` | `csp/config.py:54, 68` | +| `FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT` | 543-547 | conditional | `csp/config.py:52, 66` | +| `FACILITY_S3_BUCKET` | 660 | `""` | `csp/config.py:43`; `facility.py:211`; `users/models.py:206` | +| `FACILITY_S3_REGION` | 661 | `BUCKET_REGION` | `csp/config.py:34, 47, 61` | +| `FACILITY_S3_KEY` | 662 | `BUCKET_KEY` | `csp/config.py:36, 49, 63` | +| `FACILITY_S3_SECRET` | 663 | `BUCKET_SECRET` | `csp/config.py:37, 50, 64` | +| `FACILITY_S3_BUCKET_ENDPOINT` | 664-666 | `BUCKET_ENDPOINT` | `csp/config.py:41` | +| `FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT` | 667-671 | conditional | `csp/config.py:39`; `facility.py:211`; `users/models.py:206` | + +**verified** `BUCKET_PROVIDER` is validated against `CSProvider.__members__` at +`base.py:533-534`, but an invalid value only logs an error — it does not raise. + +**verified** `CSProvider` (`csp/config.py:17-24`) already includes a `GCP` member +at line 20. **verified** `BUCKET_PROVIDER` is only ever compared against +`CSProvider.AWS_ROLE_BASED` (`csp/config.py:35, 48, 62`); the `GCP` member is +never branched on anywhere in the repository. + +--- + +## 8. Replaceability assessment + +| Category | Call sites | Django Storage API verdict | +| --- | --- | --- | +| Plain read/write/delete | 3, 4, 5, 6, 8, 12, 17, 18 | **Replaceable.** `save()` / `open()` / `delete()` cover these. | +| Batch delete | 7 | Dead code; delete it or loop `Storage.delete()`. | +| ACL-tagged write | 9 | Needs a decision on public objects under GCS. | +| Presigned URL generation | 1, 2, 13-16, 19 | **Not replaceable.** Must become Django-served routes to meet the stated goal. | +| Unsigned URL construction | `facility.py:211`, `users/models.py:206` | **Not replaceable** as-is; depends on public objects. | + +**inferred** 8 of 19 call sites port mechanically. The remaining 11 are the actual +migration work, and 9 of those exist only to hand object-storage URLs to the +browser — which the target architecture forbids. + +--- + +## 11. IS-01 migration status + +Recorded 2026-08-06 on `feature/django-storages`, then revised. The prediction in +§8 held for the mechanical work: the 8 mechanical sites migrated and the 2 dead +ones were deleted. It was wrong about the 9 URL-handing sites, which it expected +to survive into IS-02 — the ES-01 completion pass removed the signed-URL +transport outright instead, and ES-02 later removed the base64 body. The table +below is the current status, not the prediction. + +### 11.1 Status of every call site + +Numbering follows the §3 table. + +Final status, after the ES-01 completion pass removed the signed-URL transport. + +| # | Symbol | Status | +| --- | --- | --- | +| 1 | `signed_url` | **removed** — presigned PUT deleted; uploads use the Django endpoint | +| 2 | `read_signed_url` | **removed** — replaced by `GET /files/{id}/download/` | +| 3 | `put_object` | `migrated_to_django_storage` — `Storage.save()` | +| 4 | `get_object` | `migrated_to_django_storage` — `Storage.open()` | +| 5 | `file_contents` | `migrated_to_django_storage` — `Storage.open().read()`; still no caller | +| 6 | `delete_object` | `migrated_to_django_storage` — `Storage.delete()` | +| 7 | `delete_objects` | **removed** — dead code, and no portable batch delete exists | +| 8 | `delete_cover_image` | `migrated_to_django_storage` — `storages["facility"].delete()` | +| 9 | `upload_cover_image` | `migrated_to_django_storage` — `storages["facility"].save()`, no ACL | +| 10 | `FileUpload.files_manager` | `temporary_wrapper` — `FilesManager("patient")` | +| 11 | `ReportUpload.files_manager` | `temporary_wrapper` — `FilesManager("report")` | +| 12 | `FileUploadViewSet.upload_file` | `migrated_to_django_storage` — persistence *and* transport; multipart since ES-02 | +| 13-16 | signed/read URL fields on both retrieve specs | **removed** — replaced by `download_url`, a CARE route | +| 17 | `generate_and_upload_report` | `migrated_to_django_storage` — `Storage.save()` with `ContentFile` | +| 18 | `cleanup_incomplete_file_uploads` | `migrated_to_django_storage` — `Storage.delete()` | +| 19 | report data point `read_signed_url` | **removed** — embeds a CARE download route instead | +| — | `facility.py`, `users/models.py` URL builders | `migrated_to_django_storage` — now reverse to CARE asset routes | + +Totals over all **21** call sites — the 19 numbered in §3 plus the two +unnumbered URL builders. Two rows above cover more than one site: `13-16` is +four fields, and the last row is two builders. + +| Status | Sites | Which | +| --- | --- | --- | +| `migrated_to_django_storage` | **11** | 3, 4, 5, 6, 8, 9, 12, 17, 18, and both URL builders | +| `temporary_wrapper` | **2** | 10, 11 — the two `files_manager` bindings | +| **removed** | **8** | 1, 2, 7, 13, 14, 15, 16, 19 | + +Nothing is `legacy_signed_url_only`. Nothing is `blocked`. + +### 11.1a Transport now mediated by CARE + +| Route | Serves | Auth | +| --- | --- | --- | +| `GET /api/v1/files/{external_id}/download/` | patient files | `file_authorizer` via `get_queryset` | +| `GET /api/v1/template_reports/{external_id}/download/` | reports | `read_report_authorizer`, called explicitly | +| `GET /api/v1/assets/facility/{external_id}/cover_image/` | cover images | anonymous | +| `GET /api/v1/assets/user/{username}/profile_picture/` | avatars | anonymous | + +**verified** The two asset routes are anonymous by design. `AllFacilityViewSet` +and `FacilitySchedulableUsersViewSet` are unauthenticated and already expose +these images, so who can see them is unchanged. What changed is that CARE reads +the bytes through Django Storage, which lets the bucket be private. They are +separate views because `FacilityViewSet` and `UserViewSet` filter their +querysets by `request.user` and cannot serve an anonymous request. + +**verified** All four stream via `FileResponse`; nothing buffers a whole object +to serve it. The inline-vs-attachment decision that presigned +`ResponseContentDisposition` used to make is preserved in +`care/emr/utils/file_download.py`. + +### 11.2 Architecture now in place + +**verified** Application code addresses logical aliases; the provider is chosen +in settings alone. + +| Alias | Physical bucket setting | Consumers | +| --- | --- | --- | +| `patient` | `CARE_PATIENT_STORAGE_BUCKET` (defaults to `FILE_UPLOAD_BUCKET`) | `FileUpload.files_manager` | +| `facility` | `CARE_FACILITY_STORAGE_BUCKET` (defaults to `FACILITY_S3_BUCKET`) | `care/utils/file_uploads/cover_image.py` | +| `report` | `CARE_REPORT_STORAGE_BUCKET` (defaults to `FILE_UPLOAD_BUCKET`) | `ReportUpload.files_manager` | +| `staticfiles` | — | WhiteNoise, unchanged | + +**verified** `CARE_STORAGE_BACKEND` selects `s3` (default) or `gcs`. An +unsupported value raises `ImproperlyConfigured` naming the supported values. + +**verified** `report` and `patient` still resolve to the same physical bucket, as +before, but are independently configurable. + +### 11.3 Object-name generation + +**verified** `care/emr/utils/file_manager.py:get_storage_name` is the single +provider-neutral name helper. The `/` convention is +preserved byte-for-byte; names are relative and carry no bucket, URL or endpoint. +Traversal, absolute paths and empty components raise `SuspiciousFileOperation`. + +**verified** Cover images and avatars keep their own unrelated convention, +`/_.`, and are addressed through the alias +directly rather than through `FilesManager`. + +### 11.3a One source of truth per logical bucket + +**verified** Each logical bucket is resolved in exactly one place — the +`STORAGES` alias built in `config/settings/base.py`. Nothing else derives a +bucket name. + +| Alias | Setting | Fallback | +| --- | --- | --- | +| `patient` | `CARE_PATIENT_STORAGE_BUCKET` | `FILE_UPLOAD_BUCKET` | +| `facility` | `CARE_FACILITY_STORAGE_BUCKET` | `FACILITY_S3_BUCKET` | +| `report` | `CARE_REPORT_STORAGE_BUCKET` | `FILE_UPLOAD_BUCKET` | + +**This closes a defect introduced earlier in IS-01.** While the signed-URL path +survived, it resolved buckets through `care/utils/csp/`, which read the *old* +settings. Setting `CARE_PATIENT_STORAGE_BUCKET` therefore moved persistence but +not the URLs, so uploads and downloads silently addressed different buckets — +on the default `s3` profile, not only under `gcs`. Removing the signed-URL +transport removes the second resolver, so the divergence is now structurally +impossible: `download_url` names a CARE route and carries no bucket at all. + +### 11.4 §2.1 credential defect — corrected + +**verified** The defect recorded in §2.1 is fixed. The `patient` and `report` +aliases now read `FILE_UPLOAD_REGION`, `FILE_UPLOAD_KEY` and +`FILE_UPLOAD_SECRET`; `facility` reads the `FACILITY_S3_*` set. The three +`FILE_UPLOAD_*` credential settings listed as dead in §7 are now live. + +**Behaviour change.** Locally every one of these resolves to the same MinIO +value, so there is no local change. A deployment that sets `FACILITY_S3_KEY` to +something other than `BUCKET_KEY` *without* also setting `FILE_UPLOAD_KEY` will +now use `BUCKET_KEY` for patient and report objects and must set +`FILE_UPLOAD_KEY` explicitly. + +**verified** `get_patient_bucket_config` and `get_report_bucket_config` carried +the same defect. An earlier revision of this section recorded them as left in +place for IS-02 to remove; that is no longer true. The completion pass deleted +`care/utils/csp/` outright along with the signed-URL path that was their only +caller, so there is no second resolver left to diverge from settings (§11.5). + +### 11.5 Provider SDK use remaining + +**verified** After the ES-01 completion pass, **no storage module imports a +provider SDK**. `boto3` / `botocore` appear in exactly two non-test modules, and +neither performs object storage: + +| File | Line | Use | Why retained | +| --- | --- | --- | --- | +| `care/emr/tasks/report_generation.py` | 1 | `ClientError` in `autoretry_for` | Retry configuration only; never instantiates a client. Under `s3` django-storages raises it from inside `Storage.save`, so retry is unchanged; under `gcs` it would not fire. Left alone because the completion pass forbids modifying Celery. See §11.7 | +| `care/utils/sms/backend/sns.py` | 8-9, 44, 55 | SNS client | SMS delivery, not storage | + +**verified** No module imports `google.cloud.storage` directly. GCS is reached +only through `django-storages`. + +**verified** `care/utils/csp/` is deleted. `BucketType`, `CSProvider`, +`ClientConfig` and `get_client_config` existed only to resolve provider +credentials and external endpoints for signed URLs, and have no consumer left. +`BUCKET_PROVIDER` survives as the credential-source switch, compared against a +literal in `config/storage.py`. + +**verified** Settings deleted for want of a consumer: `FACILITY_CDN`, +`BUCKET_HAS_FINE_ACL`, `BUCKET_EXTERNAL_ENDPOINT`, +`FILE_UPLOAD_BUCKET_EXTERNAL_ENDPOINT`, `FACILITY_S3_BUCKET_EXTERNAL_ENDPOINT`. + +### 11.6 Whole-file reads remaining + +Updated from §5. + +| Site | File | Status | +| --- | --- | --- | +| 5 | `file_manager.file_contents` | Retained as an explicit opt-in; still no production caller. `get_object` returns a stream and is the default. | +| 12 | `file_upload.py` base64 decode | **Gone (ES-02).** Multipart replaced it. Django's upload handlers decide memory vs temporary file; CARE reads only the leading 2048 bytes to sniff the MIME type and hands the `UploadedFile` straight to `Storage.save()`. | +| 17 | `report_utils.py` `output_bytes` | **Remains.** The renderer returns complete bytes; ES-01 §18 explicitly does not require redesigning report generation. | + +**verified** `upload_cover_image` no longer buffers: the `UploadedFile` is passed +to `Storage.save()` directly instead of `image.file` being handed to +`put_object`. + +### 11.7 Known gap — retry under the GCS profile + +**verified** `care/emr/tasks/report_generation.py:13` retries on +`botocore.exceptions.ClientError`. Under `CARE_STORAGE_BACKEND=gcs`, storage +failures raise `google.api_core.exceptions.*`, so report generation would not +retry. Generation on the happy path is unaffected; what is absent is the +recovery. A transient upload failure that `s3` would ride out fails the report +outright under `gcs`, on its first attempt, with a retry policy present in the +code that cannot fire. + +**Not changed.** Both ES-01 §31 and the completion pass forbid modifying Celery. +This is the last provider-specific reference in a storage consumer and is the +one item that should be resolved before the GCS profile is used in anger. +Recorded in `unresolved-items.md` S2; `02-target-runtime.md` §11 makes closing +it — or excluding report generation from GCS readiness claims — a requirement of +the target runtime rather than an open note. + +### 11.8 Remaining compatibility layers + +**verified** Two, both deliberate: + +| Layer | Purpose | Removal | +| --- | --- | --- | +| `FilesManager` | Binds `FileUpload` / `ReportUpload` to a logical alias. Pure Django Storage; no provider import or branch. | Optional. It is a convenience, not a portability risk. | +| `S3FilesManager` | Deprecated subclass kept so external plugins that import the old name keep working. Warns, delegates to Django Storage, exposes **no** signed-URL method, rejects unknown aliases. | After plugin authors migrate; see `plugin-impact.md`. | + +**verified** The base64 upload transport at `POST /api/v1/files/upload-file/` was +removed by ES-02 and replaced with `multipart/form-data`. No upload path buffers +a complete file any more. See `frontend-file-flow.md` §12. diff --git a/docs/xii/architecture/inventory/task-call-sites.md b/docs/xii/architecture/inventory/task-call-sites.md new file mode 100644 index 0000000000..a0420c736e --- /dev/null +++ b/docs/xii/architecture/inventory/task-call-sites.md @@ -0,0 +1,398 @@ +--- +title: Task Call-Site Inventory +document: inventory/task-call-sites +version: 0.1.0 +status: Draft +phase: 0 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +reviewed: 2026-08-05 +--- + +# Task Call-Site Inventory + +Every Celery task definition and dispatch site in the repository. No task was +migrated or modified in this phase. + +Evidence labels: **verified** / **inferred** / **unknown**. + +Classifications are restricted to the five permitted values: +`cloud_tasks_candidate`, `cloud_run_job_candidate`, `synchronous_candidate`, +`celery_compatibility`, `requires_analysis`. + +--- + +## 1. Summary + +**verified** There are **8 task definitions** in the repository. + +**verified** The most consequential finding of this section: **5 of the 8 are +decorated as Celery tasks but invoked as plain Python function calls at nearly +every production call site.** They execute inline, inside the request thread, +and never reach a broker. + +**verified** **4 tasks reach a broker** in non-test code, in two different ways: + +- **3 are dispatched asynchronously at every call site** — + `generate_report_task`, `send_totp_enabled_email`, `send_totp_disabled_email`. +- **1 is mixed** — `summarise_monetary_components` is called inline from + production code but re-dispatches *itself* asynchronously in its recursive + tail, so it reaches a broker only from inside itself. + +The distinction matters for migration: the first three can move to a task +backend by changing their dispatch, while the fourth also needs its recursive +self-dispatch re-expressed, and that path has no inline equivalent to fall back +on. + +**verified** Celery app: `config/celery_app.py`. Instantiated at +`celery_app.py:8` as `Celery("care")`, autodiscovery at `celery_app.py:18`. + +--- + +## 2. Celery application configuration + +**verified** `config/celery_app.py`: + +| Line | Fact | +| --- | --- | +| 6 | `os.environ.setdefault("DJANGO_SETTINGS_MODULE", "config.settings.production")` | +| 8 | `app = Celery("care")` | +| 14 | `app.config_from_object("django.conf:settings", namespace="CELERY")` | +| 16 | `app.conf.update(enable_utc=False, timezone="Asia/Kolkata")` | +| 18 | `app.autodiscover_tasks()` | + +**verified** `config/settings/production.py` exists (243 bytes). + +**verified** `celery_app.py:16` **hardcodes** `timezone="Asia/Kolkata"` and +`enable_utc=False`. This overrides the `CELERY_TIMEZONE` derived from +`TIME_ZONE` at `config/settings/base.py:417-419`, because `app.conf.update` runs +after `config_from_object`. Any schedule expressed in `crontab()` is therefore +interpreted in IST regardless of deployment region. + +**verified** Broker and result backend, `config/settings/base.py`: + +| Setting | Line | Value | +| --- | --- | --- | +| `CELERY_BROKER_URL` | 421 | `env("CELERY_BROKER_URL", default=REDIS_URL)` | +| `CELERY_RESULT_BACKEND` | 423 | `= CELERY_BROKER_URL` (no independent env var) | +| `CELERY_TASK_TIME_LIMIT` | 432 | `1800 * 5` = 9000 s | +| `CELERY_TASK_SOFT_TIME_LIMIT` | 435 | `1800` s | +| `CELERY_ACCEPT_CONTENT` | 425 | `["json"]` | +| `CELERY_TASK_SERIALIZER` | 427 | `"json"` | + +**verified** `CELERY_RESULT_BACKEND` cannot be pointed anywhere other than the +broker without a code change — `base.py:423` assigns it directly from +`CELERY_BROKER_URL`. + +**verified** No `AsyncResult` usage exists anywhere in the repository. A grep for +`AsyncResult` returns zero matches in `care/` and `config/`. + +**verified** No `.apply_async(` and no `send_task(` call sites exist. A grep for +both returns zero matches. + +**inferred** Because no caller ever reads a result or a task ID, the result +backend is written but never consumed. This is the single strongest argument that +the result backend can be dropped rather than ported. + +--- + +## 3. Task definitions + +### 3.1 `generate_report_task` + +| Field | Value | +| --- | --- | +| Source | `care/emr/tasks/report_generation.py:12-78` | +| Decorator | `@shared_task(autoretry_for=(ClientError,), retry_kwargs={"max_retries": 3}, expires=10*60)` (`:12-14`) | +| Payload | `template_id: str`, `report_type: str`, `associating_id: str`, `output_format: str = "pdf"`, `**kwargs` (`:15-21`) | +| Return value | `str(report_upload.external_id)` (`:66`) | +| Call sites | `care/emr/api/viewsets/report/report_upload.py:147` — `.delay(...)` | +| Caller uses task ID | **no** — `report_upload.py:147-157` discards the `AsyncResult` and returns HTTP 201 with an empty body | +| Caller reads result | **no** | +| Retry policy | 3 retries, only on `botocore` `ClientError` (`:13`) | +| Expiry | 600 s (`:13`) | +| Duration | **unknown** — depends on WeasyPrint render time; no instrumentation in repo | +| DB effects | reads `Template` (`:39`); creates/updates/deletes `ReportUpload` via `report_utils.generate_and_upload_report` (`report_utils.py:105-133`) | +| Storage effects | `put_object` into `REPORT` bucket (`report_utils.py:127-129`) | +| Email / external | none | +| Progress state | `report_utils.set_lock` / `clear_lock` — cache-backed, see `cache-and-redis.md` §4 | +| Idempotency | **not idempotent** — each successful run creates a new `ReportUpload` row and a new object keyed by `uuid4()` + timestamp (`report_utils.py:103`), so re-requesting a report duplicates both. A *retry* is narrower: it fires only on `ClientError` from `put_object` (`report_generation.py:13`), and that path deletes the row before re-raising (`report_utils.py:133`), so a normal retry does not leave two rows. It can leave a duplicate **object**: a write ambiguous enough to raise after the bytes landed is never cleaned up, and the retry then writes a second object under a fresh key. Orphan *rows* are possible rather than guaranteed — they need a failure between the row save and the storage write, or one ambiguous enough that the cleanup itself does not run. See the note below. | +| Periodic | no | +| Plugin-owned | no | +| **Classification** | `cloud_tasks_candidate` | + +**verified** The `expires=10*60` combined with `retry_kwargs={"max_retries": 3}` +is internally inconsistent: a task that expires after 10 minutes but retries up +to 3 times can have its retries discarded on expiry. Recorded in +`unresolved-items.md` §6. + +**verified** Failure path deletes the `ReportUpload` row (`report_utils.py:133`) +but only when `put_object` raises. If the process is killed between +`report_upload.save(skip_internal_name=True)` (`:122`) and the `put_object` +(`:127`), the row survives with `upload_completed=False` and no object. + +### 3.2 `send_totp_enabled_email` + +| Field | Value | +| --- | --- | +| Source | `care/emr/tasks/totp.py:8-32` | +| Decorator | `@shared_task(autoretry_for=(Exception,), retry_kwargs={"max_retries": 3}, expires=10*60)` (`:8-12`) | +| Payload | `user_email: str`, `user_name: str` (`:13`) | +| Return value | `None` | +| Call sites | `care/emr/api/viewsets/totp.py:102` — `.delay(user.email, user.username)` | +| Caller uses task ID / result | **no** | +| Retry policy | 3 retries on **any** `Exception` (`:9`) | +| Expiry | 600 s (`:11`) | +| Duration | **inferred** sub-second plus SMTP round trip | +| DB effects | none | +| Storage effects | none | +| Email | **yes** — `EmailMessage(...).send()` (`:25-32`) | +| Idempotency | **not idempotent** — a retry re-sends the email. `autoretry_for=(Exception,)` is broad enough that a post-send failure would duplicate delivery. | +| Periodic | no | +| Plugin-owned | no | +| **Classification** | `cloud_tasks_candidate` | + +### 3.3 `send_totp_disabled_email` + +**verified** Identical shape to §3.2. + +| Field | Value | +| --- | --- | +| Source | `care/emr/tasks/totp.py:35-59` | +| Decorator | `@shared_task(autoretry_for=(Exception,), retry_kwargs={"max_retries": 3}, expires=10*60)` (`:35-39`) | +| Payload | `user_email: str`, `user_name: str` (`:40`) | +| Call sites | `care/emr/api/viewsets/totp.py:139` — `.delay(user.email, user.username)` | +| Email | **yes** — `:52-59` | +| Idempotency | **not idempotent**, same reasoning as §3.2 | +| **Classification** | `cloud_tasks_candidate` | + +### 3.4 `cleanup_expired_token_slots` + +| Field | Value | +| --- | --- | +| Source | `care/emr/tasks/cleanup_expired_token_slots.py:12-21` | +| Decorator | `@shared_task` — bare, no retry, no expiry (`:12`) | +| Payload | none | +| Return value | `None` | +| Call sites | **periodic only** — `care/emr/tasks/__init__.py:13-17` | +| Retry policy | none | +| Expiry | none | +| Duration | **unknown** — single unbounded `queryset.delete()` (`:21`) over all expired unbooked `TokenSlot` rows | +| DB effects | **hard delete** of `TokenSlot` where `tokenbooking__isnull=True` and `end_datetime__lte=now()` (`:18-21`) | +| Storage / email | none | +| Idempotency | **idempotent** — re-running deletes nothing further | +| Periodic | `crontab(hour="0", minute="0")` — daily midnight, IST per §2 (`__init__.py:14`) | +| Plugin-owned | no | +| **Classification** | `cloud_run_job_candidate` | + +**verified** Unlike `cleanup_incomplete_file_uploads`, this task does **not** +paginate. `queryset.delete()` at `:21` issues one delete over the whole matching +set. **inferred** On a large table this can exceed a request-scoped timeout, +which is why a Job rather than a Cloud Tasks handler is the right shape. + +### 3.5 `cleanup_incomplete_file_uploads` + +| Field | Value | +| --- | --- | +| Source | `care/emr/tasks/cleanup_incomplete_file_uploads.py:14-57` | +| Decorator | `@shared_task()` — no retry, no expiry (`:14`) | +| Payload | none | +| Return value | `True` (`:57`) | +| Call sites | periodic (`__init__.py:19-24`); test `care/emr/tests/test_file_upload_api.py:177` | +| Caller reads result | **no** — the `True` return is never consumed | +| Retry policy | none | +| Expiry | none | +| Duration | **unknown** — loops in pages of 1000 (`:21`) until the queryset is empty (`:28`) | +| DB effects | hard-deletes `FileUpload` rows (`:44`) | +| Storage effects | `delete_object` per file (`:33`), `PATIENT` bucket | +| Email | none | +| Idempotency | **idempotent** in effect — `delete_object(quiet=True)` tolerates missing keys (`:33`, and `file_manager.py:106-110`) | +| Periodic | every `FILE_UPLOAD_EXPIRY_HOURS` hours, as an **interval in seconds**, not a crontab (`__init__.py:20-24`) | +| Plugin-owned | no | +| **Classification** | `cloud_run_job_candidate` | + +**verified** A defect in the loop: line 40 re-raises after logging +(`raise e` inside the `except` at `:34-40`). Because the raise happens *before* +`ids_to_delete.append(file.id)` at `:41`, one unexpected storage error aborts the +whole run and the successfully deleted objects in that page are never removed +from the database. Their rows remain with `upload_completed=False`, so the next +run retries them — self-healing, but it means a single poison object stalls +cleanup indefinitely. Recorded in `unresolved-items.md` §7. + +**verified** `quiet=True` only suppresses `s3.exceptions.NoSuchKey` +(`file_manager.py:106`). Other `ClientError`s propagate and hit the `raise` above. + +### 3.6 `summarise_monetary_components` + +| Field | Value | +| --- | --- | +| Source | `care/emr/models/resource_category.py:123-140` | +| Decorator | `@shared_task` — bare (`:123`) | +| Payload | `category: ResourceCategory \| int` (`:124`) | +| Return value | `None` | +| Retry / expiry | none | +| **Classification** | `requires_analysis` | + +**verified** Call sites — note the mixed dispatch: + +| File | Line | Dispatch | +| --- | --- | --- | +| `care/emr/models/resource_category.py` | 91 | `summarise_monetary_components(self)` — **synchronous**, passes a **model instance** | +| `care/emr/models/resource_category.py` | 140 | `summarise_monetary_components.delay(component.id)` — **async**, passes an **int** | +| `care/emr/api/viewsets/resource_category.py` | 208 | `summarise_monetary_components(obj.id)` — **synchronous**, passes an **int** | + +**verified** The `isinstance(category, int)` branch at `:125-126` exists precisely +because the parameter is sometimes a model instance and sometimes a primary key. + +**verified** This task is **self-recursive and fans out**: line 140 dispatches one +async task per child category, and each child repeats. Depth and width are +bounded only by the `ResourceCategory` tree. + +**verified** A `ResourceCategory` model instance is **not JSON-serializable**, and +`CELERY_TASK_SERIALIZER = "json"` (`base.py:427`). The `:91` call site would fail +if it were ever dispatched with `.delay()`. It works only because it is called +synchronously. + +**inferred** This is why the classification is `requires_analysis`: converting the +synchronous call sites to Cloud Tasks changes a currently-transactional in-request +mutation into an eventually-consistent fan-out, and the recursive dispatch would +multiply Cloud Tasks invocations by the size of the category tree. + +### 3.7 `handle_cascade` + +| Field | Value | +| --- | --- | +| Source | `care/emr/models/location.py:159-167` | +| Decorator | `@app.task` — bare, uses the app object directly, imported at `location.py:9` (`from config.celery_app import app`) | +| Payload | `base_location` — a `FacilityLocation` **primary key** (`:160`, per the caller at `:128`) | +| Return value | `None` | +| Retry / expiry | none | +| DB effects | re-saves every descendant `FacilityLocation` with `update_fields=["cached_parent_json"]` (`:166`) | +| Storage / email | none | +| Periodic | no | +| Plugin-owned | no | +| **Classification** | `requires_analysis` | + +**verified** Both call sites are **synchronous**: + +- `care/emr/models/location.py:128` — `FacilityLocation.cascade_changes` calls + `handle_cascade(self.id)` with no `.delay()`. +- `care/emr/models/location.py:167` — the task **recurses into itself + synchronously**: `handle_cascade(child)`. + +**verified** The recursion at `:167` passes `child`, a **`FacilityLocation` +instance**, while the entry point at `:128` passes `self.id`, an **int**. The +function body at `:165` uses the argument as `parent_id=base_location`, which +works for an int; passing a model instance relies on Django coercing the instance +to its PK in the filter. Inconsistent, but functional today. + +**inferred** Since the recursion is a direct call rather than `.delay()`, the +entire subtree is walked in one synchronous pass inside whatever request or task +triggered it. Depth is bounded by the location hierarchy; Python's recursion +limit applies. + +### 3.8 `rebalance_account_task` + +| Field | Value | +| --- | --- | +| Source | `care/emr/resources/account/sync_items.py:81-85` | +| Decorator | `@app.task()` — imported at `sync_items.py:14` | +| Payload | `account_id` — an `Account` primary key (`:82`) | +| Return value | `None` | +| Retry / expiry | none | +| DB effects | `Account.objects.get` (`:83`), `sync_account_items(account)` (`:84`), `account.save()` (`:85`) | +| Storage / email | none | +| Periodic | no | +| Plugin-owned | no | +| **Classification** | `requires_analysis` | + +**verified** **Every one of its 12 non-test call sites is synchronous.** Not one +uses `.delay()`: + +| File | Lines | +| --- | --- | +| `care/emr/api/viewsets/charge_item.py` | 456 | +| `care/emr/api/viewsets/invoice.py` | 160, 224, 257, 287, 308 | +| `care/emr/api/viewsets/payment_reconciliation.py` | 95, 115, 178, 229 | +| `care/emr/resources/invoice/return_items_invoice.py` | 87, 119 | +| `care/emr/tests/test_payment_reconciliation_api.py` | 221 (test) | + +**inferred** This task is financial-balance recalculation running inline in the +request path. Making it asynchronous would expose intermediate inconsistent +balances to reads. The `@app.task` decorator appears to be aspirational rather +than active — nothing in this repository dispatches it to a worker. + +--- + +## 4. Periodic schedule + +**verified** `care/emr/tasks/__init__.py:11-24`, registered on the +`current_app.on_after_finalize` signal: + +| Task | Schedule | Line | Notes | +| --- | --- | --- | --- | +| `cleanup_expired_token_slots` | `crontab(hour="0", minute="0")` | 13-17 | daily 00:00, **IST** per §2 | +| `cleanup_incomplete_file_uploads` | `FILE_UPLOAD_EXPIRY_HOURS * 3600` seconds | 20-24 | interval, not crontab | + +**verified** The second registration is conditional: +`if cleanup_file_upload_hours := settings.FILE_UPLOAD_EXPIRY_HOURS:` +(`__init__.py:19`). Setting `FILE_UPLOAD_EXPIRY_HOURS=0` disables the schedule +entirely. Default is `24` (`config/settings/base.py:720`). + +**verified** The schedule is defined **in code**, not in a database scheduler. +There is no `django-celery-beat` dependency in `Pipfile`. The beat process holds +the schedule in memory and persists only its own timing file. + +**inferred** Both schedules map cleanly to Cloud Scheduler triggers. The interval +form (`cleanup_incomplete_file_uploads`) has no anchor, so its first fire is +relative to beat startup; a Cloud Scheduler cron equivalent would need an +explicit time chosen. + +--- + +## 5. Classification roll-up + +| Classification | Count | Tasks | +| --- | --- | --- | +| `cloud_tasks_candidate` | 3 | `generate_report_task`, `send_totp_enabled_email`, `send_totp_disabled_email` | +| `cloud_run_job_candidate` | 2 | `cleanup_expired_token_slots`, `cleanup_incomplete_file_uploads` | +| `requires_analysis` | 3 | `summarise_monetary_components`, `handle_cascade`, `rebalance_account_task` | +| `synchronous_candidate` | 0 | — | +| `celery_compatibility` | 0 | — | + +**Note on the empty `synchronous_candidate` row:** the three +`requires_analysis` tasks are *already* synchronous at their call sites. They are +not candidates to *become* synchronous — they are candidates to have their +misleading task decorators either removed or actually used. Classifying them +`synchronous_candidate` would imply a change that is already the status quo, so +the analysis label is the honest one. + +--- + +## 6. Dispatch-mechanism summary + +**verified**: + +| Mechanism | Count | Sites | +| --- | --- | --- | +| `.delay(` | 5 | `report_upload.py:147`, `totp.py:102`, `totp.py:139`, `resource_category.py:140`, `test_file_upload_api.py:177` (test) | +| `.apply_async(` | 0 | — | +| `send_task(` | 0 | — | +| `AsyncResult` | 0 | — | +| `add_periodic_task` | 2 | `tasks/__init__.py:13`, `tasks/__init__.py:20` | +| `crontab` | 1 | `tasks/__init__.py:14` | +| Synchronous calls to `@task`-decorated functions | 17 (16 non-test) | §3.6, §3.7, §3.8 | + +Breakdown of the 16 non-test synchronous calls: +`rebalance_account_task` 12, `summarise_monetary_components` 2 +(`resource_category.py:91`, `viewsets/resource_category.py:208`), +`handle_cascade` 2 (`location.py:128`, `location.py:167`). + +**verified** In non-test production code, asynchronous dispatch happens at +exactly **4 call sites**. Synchronous invocation of task-decorated functions +happens at **16** — four times as often. + +**inferred** The practical migration surface for Cloud Tasks is therefore much +smaller than the count of `@shared_task` decorators suggests: 4 dispatch sites +across 4 distinct tasks, plus 2 scheduled jobs. diff --git a/docs/xii/architecture/inventory/unresolved-items.md b/docs/xii/architecture/inventory/unresolved-items.md new file mode 100644 index 0000000000..1ca7dcbef6 --- /dev/null +++ b/docs/xii/architecture/inventory/unresolved-items.md @@ -0,0 +1,547 @@ +--- +title: Unresolved Items +document: inventory/unresolved-items +version: 0.2.0 +status: Draft +phase: 0 +source_repository: https://github.com/ohcnetwork/care +source_branch: gcp +source_commit: 6a2976dc2512c2c532fcc70628c5690fbbbe3f3d +baseline_commit: 2fe40cd16 +reviewed: 2026-08-06 +--- + +# Unresolved Items + +Open questions, code defects found while inventorying, and contradictions between +the existing GCP documents and the verified state of the repository. + +Nothing here was fixed in Phase 0. Each item states what is **verified**, what is +**inferred**, and what remains **unknown**. + +--- + +## Part A — Contradictions with existing documents + +Per the Phase 0 brief, the other GCP documents were not rewritten. Where a +verified code fact contradicts them, it is recorded here instead. + +### A1. Document paths are inconsistent across the set + +**verified** The eight architecture documents live at +`docs/xii/architecture/`. They were committed there in `e280e0f09`. + +**verified** `01-current-runtime.md` referenced two *different* wrong paths for +the same target document: + +| Line (before correction) | Text | +| --- | --- | +| 1933 | `docs/xii/architecture/02-target-runtime.md` | +| 1945 | `docs/xii/architecture/02-target-runtime.md` | + +**Corrected** in this phase — both now read `docs/xii/architecture/02-target-runtime.md`. + +**unknown** Whether the other seven documents contain the same wrong paths. Not +audited; only `01-current-runtime.md` was in scope. **Recommend** a path sweep +across all eight before they are published. + +### A2. `01-current-runtime.md` was wrapped in a broken code fence + +**verified** Before correction the file opened with a stray ```` ````markdown ```` +at line 1 and closed the fence at line 35 with four backticks, where three were +required to close the inner `text` block. Two further stray ` ``` ` lines sat at +the end of the file. + +**verified consequence** The YAML frontmatter and all of sections 1-2 rendered as +literal code rather than as document content, and the entire tail of the file sat +inside an unterminated block. + +**Corrected** in this phase. + +### A3. `01-current-runtime.md` §38 misstated patient-bucket credentials + +**verified** The document listed `FILE_UPLOAD_REGION`, `FILE_UPLOAD_KEY` and +`FILE_UPLOAD_SECRET` as the settings patient files use. + +**verified** They are not. `get_patient_bucket_config` +(`care/utils/csp/config.py:46-56`) reads `FACILITY_S3_REGION`, `FACILITY_S3_KEY` +and `FACILITY_S3_SECRET`. The three `FILE_UPLOAD_*` credential settings +(`config/settings/base.py:537-539`) are read by no code in the repository. + +**Corrected** in this phase. See also §2 below for the underlying defect. + +### A4. `01-current-runtime.md` §26 omitted three task definitions + +**verified** The document inventoried `care/emr/tasks/` accurately but did not +mention the three task-decorated functions defined elsewhere: +`handle_cascade` (`care/emr/models/location.py:159`), +`summarise_monetary_components` (`care/emr/models/resource_category.py:123`), +and `rebalance_account_task` (`care/emr/resources/account/sync_items.py:81`). + +**Corrected** in this phase with a descriptive addition only. + +### A5. The stated goal "keep Redis optional" is not currently supportable + +**verified** Three hard couplings prevent it, detailed in `cache-and-redis.md` §1: +`cache.set(..., nx=True)` (`care/utils/lock.py:18, 44`), +`cache.delete_pattern(...)` (`care/emr/resources/base.py:313, 315`), and +`get_redis_connection("default")` (`care/emr/models/valueset.py:77`). + +**inferred** This does not contradict the *goal*, but it does contradict any +document that treats Redis removal as configuration. It is schema and code work. + +**unknown** Whether `02-target-runtime.md` or `03-migration-plan.md` make that +assumption. Not audited. + +### A6. "All uploads pass through Django" is partly already true + +**verified** `POST /api/v1/files/upload-file/` +(`care/emr/api/viewsets/file_upload.py:213-270`) already proxies uploads through +Django as base64. + +**inferred** Any document describing the Django-proxied upload as new work should +account for this endpoint — the task is to replace a base64 path with a streaming +one and to remove the presigned alternative, not to build from nothing. + +--- + +## Part B — Code defects found during inventory + +These are pre-existing upstream issues, not regressions. None was fixed. + +### B1. Patient and report buckets use facility credentials + +**verified** `care/utils/csp/config.py:46-56` and `:59-70` set +`aws_access_key_id` / `aws_secret_access_key` from `FACILITY_S3_KEY` / +`FACILITY_S3_SECRET` while returning `settings.FILE_UPLOAD_BUCKET` as the bucket. + +**verified** `FILE_UPLOAD_REGION`, `FILE_UPLOAD_KEY`, `FILE_UPLOAD_SECRET` +(`config/settings/base.py:537-539`) are dead settings. + +**Impact (inferred):** per-bucket credential separation is impossible today. A +GCP design assuming distinct service accounts or HMAC keys per bucket must fix +this first. Note the *endpoint* settings are wired correctly, so the bug is +invisible in single-credential local and MinIO setups — which is likely why it +has survived. + +**unknown** Whether this is intentional consolidation or an unnoticed +copy-paste. **Recommend** raising upstream before diverging. + +**Partly resolved in IS-01.** The `patient` and `report` storage aliases now read +`FILE_UPLOAD_REGION`, `FILE_UPLOAD_KEY` and `FILE_UPLOAD_SECRET`, so those three +settings are no longer dead and per-bucket credentials are configurable. +`get_patient_bucket_config` and `get_report_bucket_config` still contain the +original defect, but are now reached only by the legacy signed-URL path, which +IS-02 removes. See `storage-call-sites.md` §11.4 for the behaviour change. + +### B2. LocMem/Dummy cache shims silently disable locking + +**verified** `config/caches.py:6-9` and `:12-16` accept the `nx` kwarg, ignore it, +and unconditionally `return True`. + +**verified** `Lock.acquire` (`care/utils/lock.py:17-19`) treats any truthy return +as success. Under either shim, **no lock is ever contended**. + +**verified** The shims are referenced only from `care/utils/tests/test_utils.py:18`. + +**Impact (inferred):** the most dangerous item in this document. Substituting a +PostgreSQL or LocMem cache backend does not degrade locking — it removes it, +silently, with call sites that still read as correct. Any PostgreSQL lock must be +a real conditional write (`INSERT ... ON CONFLICT DO NOTHING` or +`pg_try_advisory_lock`). + +### B3. Redis outage accepts revoked JWTs + +**verified** `config/settings/base.py:93` sets `IGNORE_EXCEPTIONS: True`. + +**verified** `config/authentication.py:21` treats a cache miss as +"token not invalidated". + +**Impact (inferred):** with Redis unreachable, `cache.get` returns `None` and +revoked access tokens are honored. This is a security property degrading open. +Contrast with `Lock.acquire`, which degrades closed under the same condition. + +**unknown** Whether this is a known accepted risk upstream. + +### B4. `cleanup_incomplete_file_uploads` aborts the batch on one storage error + +**verified** `care/emr/tasks/cleanup_incomplete_file_uploads.py:34-40` logs and +then re-raises inside the per-file loop, **before** +`ids_to_delete.append(file.id)` at line 41. + +**verified** `quiet=True` only suppresses `NoSuchKey` +(`care/emr/utils/file_manager.py:106`); any other `ClientError` propagates. + +**Impact (inferred):** one undeletable object stalls the entire cleanup +indefinitely. Rows already deleted from storage in that page are never removed +from the database, so the next run retries them — self-healing but non-progressing. + +### B5. Report generation is not idempotent under retry + +**verified** `care/emr/tasks/report_generation.py:12-14` declares +`autoretry_for=(ClientError,)` with `max_retries: 3`. + +**verified** `care/emr/reports/report_utils.py:103, 105-122` creates a **new** +`ReportUpload` row and a new object key on every invocation. + +**Impact (inferred):** each *successful* invocation produces an additional row +and stored object. A retry is narrower: the failure path deletes the row when +`put_object` raises (`report_utils.py:132-134`), so a retry does not accumulate +rows. It can accumulate objects — a write ambiguous enough to raise after the +bytes landed is not cleaned up, and the retry writes another under a fresh key. +A crash between the row save (`:122`) and the `put_object` (`:127`) leaves an +orphan row. + +### B6. `expires` and `max_retries` interact badly + +**verified** `report_generation.py:13` and `totp.py:11, 38` all set +`expires=10 * 60` alongside `max_retries: 3`. + +**inferred** A task that expires 10 minutes after dispatch can have queued +retries discarded on expiry, so the effective retry count is less than 3 whenever +the queue is backed up. Not verified against Celery's exact expiry semantics for +retried tasks. **unknown** whether the interaction was considered. + +### B7. TOTP emails retry on any exception, including post-send failures + +**verified** `care/emr/tasks/totp.py:9` and `:36` use +`autoretry_for=(Exception,)`. + +**verified** `msg.send()` is the last statement (`:32`, `:59`). + +**inferred** A failure after the SMTP handoff but before task completion +re-sends the email. Low severity, but relevant if Cloud Tasks changes retry +timing. + +### B8. Storage writes are not covered by the surrounding transaction + +**verified** `care/emr/api/viewsets/file_upload.py:255-268` wraps the model save +and `put_object` in one `transaction.atomic()` block. + +**inferred** Object storage is not transactional. A commit failure after a +successful `put_object` orphans the object; the DB row rolls back but the bytes +remain. `cleanup_incomplete_file_uploads` does not catch these, because it keys +off `FileUpload` rows — and the row no longer exists. + +### B9. `mark_upload_completed` trusts the client + +**verified** `care/emr/api/viewsets/file_upload.py:177-184` sets +`upload_completed = True` with no check that an object exists in the bucket. + +**inferred** Inherent to the presigned-PUT design: Django never observes the +upload. A Django-proxied upload path removes this class of problem entirely. + +### B10. `delete_objects` is dead code — RESOLVED (IS-01) + +**verified** `care/emr/utils/file_manager.py:112-133` had no caller and carried a +GCP-specific `NotImplemented` branch. + +**Removed in IS-01.** It was deleted rather than ported: it had no caller, and +ES-01 §17 forbids provider-specific batch calls. Django Storage defines no +portable bulk delete; a future caller should iterate `Storage.delete()`. + +### B11. Celery beat health check is a liveness lie + +**verified** `scripts/celery_beat.sh` runs `touch /tmp/healthy` **before** +exec'ing `celery beat`. + +**verified** `scripts/healthcheck.sh` probes the beat role with `ls /tmp/healthy`. + +**inferred** The marker persists after beat dies, so the container reports healthy +while scheduling nothing. Low relevance under Cloud Run (no beat), but it means +the current runtime may have silently failing schedules. + +### B12. Dead cache import shadowed by a local variable + +**verified** `care/facility/models/facility.py:4` imports +`from django.core.cache import cache`, never calls it, and binds a local +`cache = []` at line 226 inside `sync_cache`. + +**Impact:** cosmetic. Recorded because it produces a false positive in any +cache-usage grep. + +--- + +## Part B2 — Storage issues open after IS-01 + +Recorded 2026-08-06. Only issues that remain genuinely unresolved after the +storage seam moved onto Django Storage. Full detail in +`storage-call-sites.md` §11. + +### S1. The GCS profile cannot serve files end to end — RESOLVED (2026-08-07) + +**Was:** `care/emr/utils/legacy_signed_urls.py` constructed a boto3 client +directly and was S3-only, so `CARE_STORAGE_BACKEND=gcs` configured persistence +against Google Cloud Storage while both signed-URL flows silently kept pointing +at S3/MinIO — and at the *old* bucket names, since they resolved buckets through +the now-deleted `care/utils/csp/`. + +**Resolved** by removing the signed-URL transport outright rather than porting +it. CARE now serves every object through Django Storage, so `download_url` +carries neither a provider nor a bucket and both profiles behave identically. +Verified under `gcs`: persistence resolves to `GoogleCloudStorage` and +`download_url` is still `/api/v1/files/{id}/download/`. + +**Consequence:** IS-02 is no longer a prerequisite for a GCS deployment. Only S2 +below stands between the `gcs` profile and production use — and until S2 is +closed, *report generation* SHALL NOT be described as production-ready under +`gcs`, even though the rest of the file surface is. See +`02-target-runtime.md` §11. + +### S2. Report generation does not retry under GCS + +**verified** `care/emr/tasks/report_generation.py:13` uses +`autoretry_for=(ClientError,)`. Under `s3` this still works, because +django-storages raises `botocore` errors from inside `Storage.save`. Under `gcs` +the failures are `google.api_core.exceptions.*` and no retry occurs. + +**Impact (verified):** it does not degrade to "retries less often" — it degrades +to **no retry at all**, silently. The task carries a retry policy that cannot +fire, so the first transient upload failure fails the report, and nothing in the +logs distinguishes that from a policy that fired and exhausted itself. + +**Not changed** — both ES-01 §31 and the completion pass explicitly forbid +modifying Celery, and widening `autoretry_for` alters retry semantics beyond the +storage seam. `02-target-runtime.md` §11 records the two acceptable resolutions +and forbids claiming GCS report generation is production-ready until one lands. + +**Now the only item blocking the GCS profile**, and the last provider-specific +reference in any storage consumer. **Decision needed:** a provider-neutral retry +predicate, or an explicit translation at the storage boundary. Report generation +still succeeds under `gcs`; only retry-on-transient-failure is absent. + +### S3. Overwrite safety depends on a backend option, not on Django Storage + +**verified** `Storage.save()` renames on collision unless the backend is +configured otherwise; `InMemoryStorage` demonstrably does. CARE relies on +overwrite semantics, which are supplied by `file_overwrite: True` on each alias +in `config/storage.py`. + +**inferred** Any future alias, or any backend swapped in for testing, must set it +or CARE will silently write to a renamed object while the database keeps the +original `internal_name`. Asserted in `care/utils/tests/test_storage_config.py`. + +### S4. Still true after IS-01, unchanged by it + +These were recorded in Part B and remain accurate; IS-01 changed the persistence +call underneath them but not the behaviour: + +| # | Item | Note | +| --- | --- | --- | +| B4 | `cleanup_incomplete_file_uploads` aborts the batch on one storage error | Semantics preserved deliberately, per ES-01 §17 | +| B5 | Report generation is not idempotent under retry | Untouched | +| B8 | Storage writes are not covered by the surrounding transaction | Untouched; still orphans objects on commit failure | +| B9 | `mark_upload_completed` trusts the client | **Largely defused.** No client can write to the bucket any more, so a file marked complete without one is now only a bookkeeping inconsistency, not an unverified external write. The endpoint is redundant; IS-02 decides its fate | + +--- + +## Part C — Open questions requiring a decision + +### C1. Where do migrations run under Cloud Run? + +**verified** Today, `migrate` runs only in `scripts/celery_beat.sh` and +`scripts/celery-dev.sh`. The API containers never migrate. +`Procfile:2` defines a `release` phase, but no Docker path uses it. + +**Decision needed.** Cloud Run Job, deploy step, or an init container. Must also +cover `sync_permissions_roles` and `sync_valueset`, which run in the same scripts +— and the first depends on the Redis lock from B2. + +### C2. Do cover images and avatars remain public? + +**verified** Written with `ACL: public-read` when `BUCKET_HAS_FINE_ACL` is set +(`care/utils/file_uploads/cover_image.py:49-51`); read via unsigned concatenated +URLs (`care/facility/models/facility.py:207-212`, `care/users/models.py:202-207`). + +**RESOLVED by IS-01 — they stay publicly *readable*, but the bucket does not.** +No object carries an ACL any more. The bytes are served by CARE through two +anonymous routes, `facility-cover-image-asset` and `user-profile-picture-asset` +(`care/emr/api/viewsets/file_assets.py`), so who can see a cover image is +unchanged while the bucket becomes private and GCS uniform bucket-level access +is satisfied. + +Both URL builders now `reverse()` to those routes. `FACILITY_CDN` and +`BUCKET_HAS_FINE_ACL` were deleted rather than redefined: with CARE serving the +bytes, a CDN belongs in front of CARE, which the long-lived `Cache-Control` on +those responses allows. + +### C3. Is the API allowed to start without Redis? + +**verified** `scripts/start.sh` and `scripts/start-dev.sh` both call +`wait_for_redis.sh`. + +**Decision needed.** Retaining the wait makes Redis a hard Cloud Run cold-start +dependency. Removing it contradicts `IGNORE_EXCEPTIONS: True` only in spirit — +but see B3 for the security consequence of tolerating a missing Redis. + +### C4. Does the Celery result backend get ported at all? + +**verified** `CELERY_RESULT_BACKEND = CELERY_BROKER_URL` +(`config/settings/base.py:423`); no `AsyncResult` anywhere; no caller reads a +task result or ID. + +**inferred** It can be dropped. Flagged because dropping it is cheap and removes +a Redis dependency outright. + +### C5. What replaces the Celery queue-length health check? + +**verified** `config/settings/base.py:458-466` constructs +`DjangoCeleryQueueLengthHealthCheck` with `broker=REDIS_URL`. + +**Decision needed.** Under Cloud Tasks there is no Redis queue. Left as-is, the +health endpoint reports unhealthy in the target runtime. + +### C6. Email delivery on GCP + +**verified** `Pipfile` installs `django-anymail` with the `amazon-ses` extra. + +**Decision needed.** GCP has no SES equivalent. Options are keeping SES +cross-cloud, or switching provider — which changes `EMAIL_BACKEND` and the +Anymail extra. + +### C7. Celery's hardcoded `Asia/Kolkata` timezone + +**verified** `config/celery_app.py:16` sets `enable_utc=False` and +`timezone="Asia/Kolkata"`, overriding `CELERY_TIMEZONE` +(`config/settings/base.py:417-419`). + +**Decision needed.** `crontab(hour="0", minute="0")` +(`care/emr/tasks/__init__.py:14`) means midnight IST. Cloud Scheduler needs that +made explicit rather than inherited. + +### C8. `ADDITIONAL_PLUGS` must match at build and deploy + +**verified** Consumed at image build (`docker/prod.Dockerfile:39`) to `pip install`, +and again at every process start (`config/settings/base.py:19`) to populate +`INSTALLED_APPS`. + +**verified** Invalid JSON is logged and swallowed (`plugs/manager.py:27-28`). + +**Decision needed.** A mismatch yields `ModuleNotFoundError` at startup; a typo +yields silent plugin loss. Both warrant an explicit startup assertion. + +### C9. `internal_name` exposure + +**verified** `care/emr/resources/file_upload/spec.py:108` returns +`internal_name` — the storage object key — carrying the in-source comment +`# Not sure if this needs to be returned`. + +**Decision needed.** Low risk while the bucket is private; unnecessary surface +either way. + +### C10. Base64 upload endpoint is missing from the OpenAPI schema + +**RESOLVED by ES-02.** The base64 endpoint no longer exists. Its replacement at +`care/emr/api/viewsets/file_upload.py:294` carries an explicit +`@extend_schema` declaring `request={"multipart/form-data": +FileUploadMultipartSerializer}` and `responses={200: FileUploadRetrieveSpec}`, +so the upload body is now part of the generated schema and discoverable by +schema-generated clients. + +Deferring the annotation until the body changed was the right order: annotating +a base64 field that was about to be deleted would have been wasted work, and the +schema now describes a contract that will not immediately move. + +Both halves of the file contract are annotated — `download` was already. + +--- + +## Part D — Unknowns not resolvable from this repository + +| # | Unknown | Why | +| --- | --- | --- | +| D1 | Which frontend components consume `signed_url` / `read_signed_url` | Frontend is a separate repository | +| D2 | Whether any client uses the base64 `upload-file` endpoint | Absent from OpenAPI; no caller here | +| D3 | Which plugins a given deployment installs | Governed by `ADDITIONAL_PLUGS`, outside version control | +| D4 | Whether known CARE plugins are GCP-compatible | No plugin source vendored | +| D5 | Meaning of `PLUGIN_CONFIGS` keys | No consumer in this repository | +| D6 | Typical file sizes and report generation duration | No instrumentation or metrics in the repository | +| D7 | Whether B1 and B3 are known upstream | Requires checking the upstream issue tracker | +| ~~D8~~ | ~~Green baseline test count and duration~~ | **Resolved 2026-08-06** — 1912 tests, 0 skipped, ~21-29 s with `--keepdb --parallel`; see `runtime-and-deployment.md` §11.9 | + +--- + +## Part E — Baseline blockers + +**RESOLVED 2026-08-06.** A green baseline exists. Full record in +`runtime-and-deployment.md` §11. + +Baseline: commit `2fe40cd16`, all five compose services healthy, 312 migrations +applied, `makemigrations --check` clean, fixtures loaded, **1912 tests, 0 skipped**, +green on 2 of 3 runs. See E7 for the single flake. + +Disposition of the blockers recorded in the previous revision: + +| # | Blocker | Disposition | +| --- | --- | --- | +| E1 | Docker daemon not running | **resolved** — Docker Desktop started; engine 29.6.2, Compose v5.3.1 | +| E2 | `make` not installed | **not a blocker** — every `Makefile` target was translated to its underlying `docker compose` command; see `runtime-and-deployment.md` §11.3 | +| E3 | No Python environment | **not applicable** — all execution happens inside the container | +| E4 | Host Python 3.14.5 vs required `==3.13.*` | **not applicable** — the image ships Python 3.13.14 | +| E5 | Redis not running | **resolved** — supplied by the compose `redis` service, healthy | +| E6 | `.env` absent | **withdrawn — the claim was wrong.** No root `.env` is required. `docker compose config` resolves with no warnings; every interpolation has a default. The real env files, `docker/.local.env` and `docker/.prebuilt.env`, are **tracked in git**. There are no `.example` variants of either. | + +### E7. Shared-Redis test isolation failures under `--parallel` + +**Scope corrected 2026-08-06 (during IS-01).** Originally recorded as a single +flaky test. It is a defect *class* affecting at least **six** tests in **two** +families, and it fires far more often than the first sample suggested. + +**verified** Root cause: `config/settings/test.py:45-56` points the cache at +Redis with a single `KEY_PREFIX = "test_"`, shared by all 16 parallel workers, +while `cache.clear()` runs in `setUp` at `care/emr/tests/test_reset_password_api.py:24` +and `care/emr/tests/test_valueset_api.py:23, 52`. A clear in one worker discards +cache state another worker is mid-way through asserting on. + +**verified** Affected tests observed failing: + +| Family | Test | Mechanism | +| --- | --- | --- | +| Rate limiting | `test_password_request_rate_limiting` | `config/ratelimit.py:9` returns the constant key `"ratelimit"`, so the counter is global and a concurrent clear resets it — `200 != 429` | +| Rate limiting | `test_password_check_rate_limiting` | same | +| Rate limiting | `test_password_confirm_rate_limiting` | same | +| Favorites | `test_add_favorite` | asserts on values read back from the shared cache (`care/emr/api/viewsets/favorites.py:40-56`) | +| Favorites | `test_remove_favorite_single` | same | +| Favorites | `test_favorite_lists_returns_list_on_first_call` | same | + +**verified** Measured on `feature/django-storages` at 1962 tests: + +| Configuration | Result | +| --- | --- | +| Full suite, serial (`--shuffle`, no `--parallel`) | **1962/1962 OK** | +| Full suite, `--parallel --shuffle`, 6 runs | 1 green, 5 with 1-2 failures | +| Only `test_favorites_api`, `test_valueset_api`, `test_reset_password_api` in parallel, 4 runs | **4/4 failed** (76 tests, no storage code involved) | + +**verified** That last row is the decisive one: the defect reproduces with the +three cache-touching modules alone, so it is independent of any other change. + +**inferred** The observed rate rose from 1-in-3 during the Phase 0 baseline to +5-in-6 here. No cache, lock or rate-limit code changed between the two. The +likeliest explanation is scheduling: more tests and slower ones (MinIO round +trips) alter how work is distributed across the 16 workers and widen the window +in which a concurrent `cache.clear()` can land. The isolated reproduction above +shows the defect does not need those tests to be present at all. + +**Not fixed.** ES-01 §31 explicitly excludes fixing rate limiting, and the +favorites half is equally out of scope. Upstream CI runs the same +`--parallel --shuffle` combination (`.github/workflows/reusable-test.yml:77`), so +it can occur there too. **unknown** whether it is known upstream. + +**Recommended fix, for whoever owns it:** give each parallel worker its own cache +namespace, e.g. derive `KEY_PREFIX` from the worker's database suffix in +`config/settings/test.py`. That removes the whole class rather than the six +symptoms. + +**inferred, separate concern:** a globally-keyed rate limit is not only a test +problem — it means the limit is shared across all callers rather than per client. + +### E8. Transient wheel corruption in the BuildKit pip cache + +**verified** The first image build failed with +`zipfile.BadZipFile: Bad CRC-32 for file '_brotli.cpython-313-x86_64-linux-gnu.so'` +during `pipenv install` at `docker/dev.Dockerfile:22`. + +**verified** Resolved by pruning only BuildKit cache mounts +(`docker builder prune --filter type=exec.cachemount`). The rebuild succeeded and +it has not recurred. **inferred** transient corruption, not a repository defect — +recorded only so the same symptom is recognised quickly if it reappears. diff --git a/docs/xii/implementation/ES-01-storage.md b/docs/xii/implementation/ES-01-storage.md new file mode 100644 index 0000000000..8f8df17c5b --- /dev/null +++ b/docs/xii/implementation/ES-01-storage.md @@ -0,0 +1,1364 @@ +# Claude Code Implementation Specification — IS-01: Portable Storage Modernization + +You are working inside a maintained fork of: + +```text +https://github.com/ohcnetwork/care +``` + +The purpose of this phase is to modernize CARE's object-storage implementation using Django's native Storage API and `django-storages`. + +This is **not** a GCP-only refactor. + +The resulting CARE application must remain portable and must support multiple storage profiles through configuration. + +The initial supported profiles are: + +```text +Local development: +Django Storage API +→ django-storages S3Storage +→ MinIO + +Generic S3-compatible deployment: +Django Storage API +→ django-storages S3Storage +→ AWS S3, MinIO or another compatible provider + +Initial GCP deployment: +Django Storage API +→ django-storages GoogleCloudStorage +→ Google Cloud Storage +``` + +Google Cloud Storage is the first managed-cloud target, but it is not the application architecture. + +The application architecture is Django Storage API. + +--- + +# 1. Current repository state + +The runtime inventory and green baseline have already been completed. + +The baseline commit is: + +```text +755e8cb20 +``` + +The baseline established: + +- Python 3.13.14; +- Django 6.0; +- Docker and Docker Compose local runtime; +- PostgreSQL, Redis, MinIO, backend and Celery services healthy; +- 312 migrations applied; +- permissions and value sets synchronized; +- fixtures loaded; +- 1,912 tests passing in successful complete runs; +- one known pre-existing parallel-test flake related to rate limiting; +- no blocking baseline issue. + +The previous inventory commit was: + +```text +2fe40cd16 +``` + +The documentation and inventories are located under: + +```text +docs/xii/architecture/ +docs/xii/architecture/inventory/ +``` + +Read these files before changing code: + +```text +docs/xii/architecture/00-scope-and-goals.md +docs/xii/architecture/01-current-runtime.md +docs/xii/architecture/02-target-runtime.md +docs/xii/architecture/03-migration-plan.md +docs/xii/architecture/04-testing.md +docs/xii/architecture/07-configuration-reference.md + +docs/xii/architecture/inventory/storage-call-sites.md +docs/xii/architecture/inventory/frontend-file-flow.md +docs/xii/architecture/inventory/runtime-and-deployment.md +docs/xii/architecture/inventory/plugin-impact.md +docs/xii/architecture/inventory/unresolved-items.md +``` + +The inventories are authoritative for the current repository commit. + +Verify every assumption against the current source before implementing it. + +--- + +# 2. Branch and repository preconditions + +The maintained integration branch is: + +```text +gcp +``` + +Create or use this feature branch: + +```text +feature/storage-modernization +``` + +Before doing any work, report: + +```bash +git status +git branch --show-current +git log -5 --oneline +git remote -v +``` + +Verify: + +- the working tree is clean; +- the current branch is `feature/storage-modernization`; +- the branch contains commit `755e8cb20`; +- `upstream` points to `https://github.com/ohcnetwork/care`; +- the branch is based on the maintained `gcp` branch. + +Do not: + +- reset; +- rebase; +- merge unrelated branches; +- force-push; +- push any commit. + +If the current branch is incorrect, stop and report it rather than modifying Git history automatically. + +--- + +# 3. Objective + +Replace CARE's custom provider-specific object persistence with Django's Storage API using `django-storages`. + +After this phase: + +1. CARE file persistence must use Django storage aliases. +2. MinIO must remain the default local storage service. +3. MinIO must be accessed through `storages.backends.s3.S3Storage`. +4. Generic S3-compatible deployments must be configurable. +5. Google Cloud Storage must be configurable through `GoogleCloudStorage`. +6. Switching providers must require configuration changes only. +7. CARE application logic must not instantiate provider clients. +8. CARE storage logic must not call `boto3` for file persistence. +9. Static files must continue using WhiteNoise. +10. Existing local Docker Compose behavior must remain functional. +11. No GCP credentials may be required for local development. +12. No direct browser-upload or file-transport redesign is required in this phase. + +This phase modernizes the **storage seam** only. + +The next phase will modernize the HTTP file transport and remove: + +- base64 uploads; +- presigned browser uploads; +- presigned browser downloads. + +--- + +# 4. Architecture + +The required architecture is: + +```text +CARE models, services and API code + | + v +Django Storage API + | + v +django.core.files.storage.storages + | + +--------------------------------+ + | | + v v +storages.backends.s3.S3Storage storages.backends.gcloud.GoogleCloudStorage + | | + v v +MinIO / AWS S3 / compatible S3 Google Cloud Storage +``` + +Application code must refer to logical storage aliases: + +```text +patient +facility +report +staticfiles +``` + +Application code must not know which provider implements an alias. + +Provider selection belongs entirely in Django settings. + +--- + +# 5. Architectural rules + +## 5.1 Django Storage is the abstraction + +Do not create a general-purpose storage framework. + +Do not introduce: + +```text +StoragePort +StorageAdapter +StorageRegistry +StorageProviderFactory +S3FilesManager plus GCSFilesManager +MinIOFilesManager +CloudStorageService +``` + +The abstraction already exists: + +```python +django.core.files.storage.Storage +``` + +Use it. + +## 5.2 Provider-neutral application code + +Application code may use: + +```python +from django.core.files.storage import storages +``` + +Application code must not contain provider branches such as: + +```python +if settings.CARE_STORAGE_BACKEND == "gcs": + ... +elif settings.CARE_STORAGE_BACKEND == "s3": + ... +``` + +Those branches belong only in settings construction. + +## 5.3 MinIO remains the local default + +The local Docker Compose profile must continue using MinIO. + +Its implementation changes from custom `boto3` operations to: + +```text +django-storages S3Storage +``` + +The local stack must start without: + +- Google credentials; +- GCP project identifiers; +- GCS buckets; +- service-account JSON; +- internet access to GCP. + +## 5.4 GCS is additive + +GCS support must be available through configuration, but it must not replace or break S3-compatible storage. + +Do not make GCS variables mandatory when S3 is selected. + +## 5.5 No direct bucket-upload redesign yet + +Do not implement multipart API uploads in this phase. + +Do not redesign frontend contracts in this phase. + +Do not add signed upload support to a custom storage backend. + +Do not reproduce the current presigned behavior inside subclasses of `S3Storage` or `GoogleCloudStorage`. + +Current signed URL behavior may be retained temporarily in isolated legacy code only when required to preserve existing tests and API compatibility until IS-02. + +--- + +# 6. Scope + +This phase includes: + +- dependency changes; +- Django storage-alias configuration; +- local MinIO through `S3Storage`; +- generic S3-compatible configuration; +- GCS configuration support; +- replacing ordinary object persistence with Django Storage operations; +- provider-neutral object-name generation; +- storage-focused tests; +- compatibility handling required to keep the current suite green; +- storage inventory updates; +- relevant documentation corrections. + +This phase does not include: + +- multipart upload endpoints; +- frontend changes; +- removal of all signed URL API contracts; +- Cloud Run; +- Cloud SQL; +- Cloud Tasks; +- Terraform; +- Redis changes; +- cache changes; +- lock changes; +- rate-limit fixes; +- Celery migration; +- PostgreSQL queue implementation; +- PostgreSQL cache implementation; +- domain-model redesign; +- repository pattern; +- broad architectural reorganization. + +--- + +# 7. Allowed modifications + +Claude MAY modify files directly related to: + +- dependency declarations and lockfiles; +- Django storage settings; +- current CARE file-management implementation; +- storage-related helpers; +- storage-related serializers or services where required; +- storage tests; +- storage configuration tests; +- local Docker settings only when necessary to preserve MinIO; +- GCP-capable settings support; +- storage inventories and architecture documentation. + +Likely areas include, but are not limited to: + +```text +config/settings/ +care/utils/csp/ +care/emr/utils/file_manager.py +storage-related CARE models or services +storage-related tests +dependency files +lockfiles +docs/xii/architecture/ +``` + +Claude SHALL NOT modify unrelated: + +```text +patient business rules +encounter business rules +facility permissions +authentication architecture +Redis locks +rate limiting +Celery dispatch +unrelated migrations +Terraform +frontend repositories outside this checkout +``` + +Any modification outside the storage surface must be explicitly justified in the final report. + +--- + +# 8. Dependency management + +Inspect the repository's actual package-management files and commands. + +Add a version of `django-storages` compatible with: + +```text +Python 3.13 +Django 6.0 +the repository's dependency resolver +``` + +Install support for: + +```text +S3 +Google Cloud Storage +``` + +Use the dependency manager's supported extras or explicit dependencies. + +Do not edit lockfiles manually. + +Do not upgrade unrelated packages. + +Do not remove `boto3` globally merely because storage no longer uses it. + +The inventory indicates that other functionality or plugins may still require AWS libraries. + +After dependency changes: + +- rebuild the application image; +- verify dependency resolution; +- verify imports; +- record the exact dependency versions selected. + +--- + +# 9. Backend-selection configuration + +Implement a narrow provider-selection setting. + +Preferred variable: + +```text +CARE_STORAGE_BACKEND +``` + +Initial supported values: + +```text +s3 +gcs +``` + +Default value: + +```text +s3 +``` + +The default must preserve existing local behavior. + +Invalid values must cause a clear configuration error listing supported values. + +Do not introduce: + +```text +IS_GCP +USE_GCP +USE_GOOGLE_STORAGE +CLOUD_PROVIDER +``` + +as storage-logic switches. + +A future filesystem test backend may remain test-only and does not need to be exposed as a normal production option in this phase. + +--- + +# 10. Logical storage aliases + +Define these aliases in Django's `STORAGES` setting: + +```text +patient +facility +report +staticfiles +``` + +Preserve WhiteNoise for: + +```text +staticfiles +``` + +The object-storage aliases must be independently configurable even if two aliases use the same physical bucket. + +For example, the `report` alias may use the same bucket as `patient`, but application code should still request: + +```python +storages["report"] +``` + +This preserves logical intent and allows future independent configuration. + +Do not make provider names part of alias names. + +Incorrect: + +```text +gcs_patient +minio_patient +s3_report +``` + +Correct: + +```text +patient +facility +report +``` + +--- + +# 11. S3-compatible profile + +When: + +```text +CARE_STORAGE_BACKEND=s3 +``` + +configure object-storage aliases using: + +```text +storages.backends.s3.S3Storage +``` + +The profile must support: + +- local MinIO; +- AWS S3; +- reasonably compatible S3 providers supported by `django-storages`. + +Settings may include, as supported by the installed version: + +```text +bucket_name +access_key +secret_key +endpoint_url +region_name +addressing_style +signature_version +querystring_auth +file_overwrite +default_acl +``` + +Use only necessary options. + +## 11.1 Existing local configuration + +Inspect the tracked: + +```text +docker/.local.env +docker/.prebuilt.env +``` + +and existing settings. + +Reuse current local values where practical. + +Avoid requiring a new local `.env`. + +Preserve: + +- current MinIO service hostname; +- current local bucket names; +- existing local credentials; +- current region compatibility; +- internal Docker endpoint behavior. + +External browser-facing MinIO endpoints may remain temporarily for legacy signed-URL compatibility, but must not become part of the new Django Storage persistence design. + +## 11.2 Generic S3 deployment + +Do not hardcode MinIO-specific behavior in application code. + +`S3Storage` settings should permit the endpoint to be absent for AWS S3. + +MinIO-specific options should be set through environment configuration or local settings. + +--- + +# 12. GCS profile + +When: + +```text +CARE_STORAGE_BACKEND=gcs +``` + +configure object-storage aliases with: + +```text +storages.backends.gcloud.GoogleCloudStorage +``` + +Use Application Default Credentials by default. + +Do not require: + +```text +GOOGLE_APPLICATION_CREDENTIALS +``` + +inside Cloud Run or other identity-aware managed environments. + +Do not require a service-account JSON file. + +Allow local integration tests to use standard Google credential mechanisms when explicitly configured. + +Required logical bucket configuration should use provider-neutral names such as: + +```text +CARE_PATIENT_STORAGE_BUCKET +CARE_FACILITY_STORAGE_BUCKET +CARE_REPORT_STORAGE_BUCKET +``` + +If compatibility with old environment variables is retained temporarily: + +- define precedence clearly; +- prefer new provider-neutral names; +- avoid printing secret values; +- document deprecation; +- do not silently combine conflicting values. + +Do not contact live GCS during ordinary local tests. + +--- + +# 13. Storage alias construction + +Avoid copying nearly identical dictionaries repeatedly when a small settings helper can construct aliases safely. + +A settings-level helper may accept: + +```text +logical bucket name +selected backend +provider options +``` + +However: + +- keep the helper inside settings or configuration code; +- do not expose it as an application storage framework; +- keep the final `STORAGES` value explicit and understandable; +- preserve the existing `staticfiles` configuration. + +The alias configuration must be easy to inspect in tests. + +--- + +# 14. Object-name generation + +Preserve the verified convention: + +```text +/ +``` + +Verify the convention against every relevant call site. + +Create one small provider-neutral helper when it removes duplicated path logic. + +Example conceptual contract: + +```python +def get_storage_name(file_object) -> str: + ... +``` + +The helper must: + +- return a relative storage name; +- not contain a bucket; +- not contain a URL; +- not contain a provider endpoint; +- normalize only what CARE actually requires; +- preserve existing names where compatibility matters. + +Test: + +- standard names; +- prefixes; +- Unicode; +- extensions; +- unusual but valid names; +- path traversal attempts. + +Do not call private `django-storages` normalization methods from application code. + +--- + +# 15. Map logical file types to aliases + +Determine from the inventory and source how CARE distinguishes: + +```text +patient +facility +report +``` + +Implement the smallest clear mapping. + +Possible approaches include: + +- model-level storage alias property; +- helper based on verified bucket type; +- explicit service selection. + +Avoid hidden global inference. + +Do not put provider names into models. + +The mapping must be tested. + +Examples of expected intent: + +```text +patient upload -> storages["patient"] +facility upload -> storages["facility"] +report output -> storages["report"] +``` + +If reports currently share patient credentials or a physical bucket, retain that physical behavior through settings while using the logical `report` alias. + +--- + +# 16. Replace ordinary persistence operations + +Replace provider-specific CARE operations with Django Storage API. + +Use operations such as: + +```python +storage.save(name, content) +storage.open(name, "rb") +storage.exists(name) +storage.delete(name) +storage.size(name) +``` + +Where the current code writes raw bytes, use an appropriate Django file wrapper such as: + +```python +from django.core.files.base import ContentFile +``` + +Only wrap bytes when necessary. + +Where the caller already provides a Django uploaded file or file-like object, pass it through without reading all bytes first. + +Use context managers for reads. + +Do not depend on provider response dictionaries. + +Do not return raw boto3 or Google SDK responses to CARE code. + +--- + +# 17. Bulk deletion + +Django Storage does not define a portable bulk-delete API. + +Replace batch provider calls with safe iteration: + +```python +for name in names: + storage.delete(name) +``` + +Preserve current cleanup semantics. + +Determine whether current behavior: + +- ignores missing objects; +- stops on first provider failure; +- continues after failure; +- reports failed names. + +Implement and test the intended behavior. + +Do not add provider-specific batch optimization during this phase. + +A future optimization may be considered only after profiling demonstrates a real bottleneck. + +--- + +# 18. Reads and memory use + +Inspect every current `file_contents` or equivalent caller. + +Classify each as: + +```text +requires bytes +can consume a file-like object +can stream +unknown +``` + +Prefer: + +```python +with storage.open(name, "rb") as file: + ... +``` + +Do not read entire objects into memory without a verified need. + +Where report generation or another internal library requires complete bytes, retain that behavior only at that call site and document it. + +This phase does not need to redesign report-generation libraries. + +Update the inventory with remaining whole-file reads. + +--- + +# 19. Missing objects and errors + +Provider-specific exceptions must not leak into migrated consumers. + +Do not build a large custom exception hierarchy. + +Use: + +- standard Django Storage behavior; +- `FileNotFoundError` where appropriate; +- existing CARE API exceptions where already defined; +- narrowly translated errors only when required. + +Preserve verified API behavior for missing objects. + +Tests must cover: + +- object exists; +- object missing; +- storage write failure; +- storage read failure; +- storage deletion failure. + +Do not expose provider credential or bucket details in user-facing errors. + +--- + +# 20. Existing `files_manager` compatibility + +The inventory identifies multiple consumers of: + +```text +files_manager +S3FilesManager +``` + +Choose the least invasive safe path. + +Preferred order: + +1. Migrate a caller directly to the relevant Django storage alias when the patch is small and clear. +2. Retain a thin compatibility wrapper only when direct migration would create excessive unrelated changes. + +A compatibility wrapper may expose existing ordinary CRUD-shaped methods. + +It must: + +- resolve a logical storage alias; +- generate a provider-neutral object name; +- delegate to Django Storage; +- contain no provider SDK imports; +- contain no provider-specific branches; +- return provider-neutral values; +- be marked transitional where applicable. + +It must not: + +- generate presigned uploads; +- generate presigned downloads through new storage subclasses; +- recreate `boto3` semantics; +- expose raw provider clients; +- become the new permanent storage abstraction. + +Signed URL legacy behavior must be separated from ordinary persistence behavior. + +--- + +# 21. Signed URL compatibility boundary + +Current CARE exposes presigned storage flows. + +The final architecture forbids direct browser-to-bucket upload and normal direct bucket download. + +That final removal belongs to IS-02. + +For IS-01: + +1. Identify every remaining signed-upload caller. +2. Identify every remaining signed-download caller. +3. Separate signed-URL behavior from ordinary CRUD. +4. Do not add signed URL methods to Django storage subclasses. +5. Do not add custom provider-specific storage backends. +6. Preserve existing behavior only as narrowly as needed to keep the current public API and tests functioning. +7. Clearly mark legacy paths and exact callers. +8. Ensure all non-signed storage persistence already uses Django Storage. +9. Update the frontend-flow inventory with what remains for IS-02. + +If current tests permit removing a signed path without frontend work, it may be removed, but do not broaden the scope merely to remove it. + +No new feature may depend on the legacy signed URL path. + +--- + +# 22. Existing base64-through-Django upload + +The inventory established that CARE already has a Django-proxied upload endpoint using base64. + +Do not redesign it in IS-01. + +Only change its underlying persistence to use Django Storage where appropriate. + +Do not convert it to multipart yet. + +Document: + +- whether it reads complete content into memory; +- which alias it uses; +- which next-phase changes are required. + +IS-02 will replace base64 transport with `multipart/form-data` and Django upload handlers. + +--- + +# 23. Static files + +Preserve: + +```text +whitenoise.storage.CompressedManifestStaticFilesStorage +``` + +Do not move static files to MinIO or GCS. + +Do not alter static URL behavior unless dependency changes require a minimal compatibility adjustment. + +Verify: + +```bash +python manage.py collectstatic --noinput +``` + +continues to work. + +--- + +# 24. File overwrite and duplicate names + +Django Storage backends may rename duplicate object names unless overwrite behavior is configured. + +Inspect CARE's current assumptions. + +Determine whether CARE expects: + +- unique generated internal names; +- overwrite; +- duplicate prevention; +- automatic renaming. + +Do not assume the same behavior across MinIO and GCS. + +Set backend options only after verifying expected semantics. + +Add tests that assert CARE-level behavior, not provider internals. + +If internal names are already unique, document that and avoid unnecessary overwrite customization. + +--- + +# 25. ACL and public-access policy + +No object-storage alias should require public objects. + +For GCS configuration: + +- prefer uniform bucket-level access; +- do not configure public-read defaults; +- do not rely on object ACLs. + +For S3-compatible configuration: + +- avoid public default ACLs; +- keep access private unless current local tests require otherwise; +- do not expose provider URLs as the new application contract. + +This phase does not implement cloud IAM or Terraform. + +It only ensures storage settings are compatible with private buckets. + +--- + +# 26. Configuration compatibility + +The greenfield cloud configuration should use provider-neutral variables. + +Local tracked settings must continue to work. + +Where practical, support existing variables temporarily. + +Do not require users to rewrite all local configuration just to preserve MinIO. + +New variable precedence should be: + +1. explicit provider-neutral variable; +2. unambiguous existing compatibility variable; +3. safe local default where already established; +4. clear error. + +If old settings contain a verified bug—for example, credentials read from an unexpected variable set—do not preserve the bug blindly. + +Correct it with tests and document the behavior change. + +Do not expand this into a broad settings rewrite. + +--- + +# 27. Test strategy + +Add focused tests before or alongside implementation. + +Follow existing CARE test conventions. + +## 27.1 Settings tests + +Test: + +```text +default backend is s3 +s3 aliases use S3Storage +gcs aliases use GoogleCloudStorage +patient alias resolves +facility alias resolves +report alias resolves +staticfiles remains WhiteNoise +invalid backend is rejected +GCS variables are not required under s3 +S3 variables are not required under gcs except compatibility as designed +``` + +Do not instantiate real GCS clients in basic settings tests if credentials would be required. + +## 27.2 Object-name tests + +Test: + +- expected key format; +- patient object; +- facility object; +- report object; +- Unicode; +- extension handling; +- path traversal or invalid path behavior. + +## 27.3 Storage behavior tests + +Test through Django Storage abstractions: + +```text +save +open +exists +delete +missing object +content preservation +duplicate-name semantics +logical alias selection +``` + +## 27.4 Compatibility wrapper tests + +If a wrapper remains, prove it delegates through Django Storage. + +Use test storage backends or mocks at the Django Storage boundary. + +Do not mock `boto3` in new persistence tests. + +## 27.5 MinIO integration tests + +Run real integration tests against the current local MinIO container. + +Verify: + +- aliases connect; +- write succeeds; +- read succeeds; +- deletion succeeds; +- missing-object behavior is controlled; +- local Docker workflow remains unchanged. + +This is a mandatory profile. + +## 27.6 GCS configuration tests + +Verify GCS aliases are constructed correctly without requiring a live GCP project. + +An optional live GCS integration test may be created with an explicit marker or skip condition. + +Ordinary local tests must not require Google credentials. + +## 27.7 Static-import verification + +Search migrated CARE storage modules for: + +```text +boto3 +botocore +google.cloud.storage +``` + +There must be no direct provider client use for migrated file persistence. + +Do not prohibit these imports globally if other verified integrations require them. + +## 27.8 Existing API regression + +Run all existing file API tests. + +Signed URL and base64 tests may remain in this phase if compatibility paths remain. + +Do not weaken tests merely to make the refactor pass. + +--- + +# 28. Baseline and full regression + +After focused tests pass, execute the official Docker-based workflow. + +Use the exact commands recorded in: + +```text +docs/xii/architecture/inventory/runtime-and-deployment.md +``` + +Because dependencies change, rebuild the image. + +Verify: + +- PostgreSQL healthy; +- Redis healthy; +- MinIO healthy; +- Celery healthy; +- backend healthy; +- migrations succeed; +- permission synchronization succeeds; +- value-set synchronization succeeds; +- fixtures succeed; +- static collection succeeds; +- complete suite runs with the same parallel and shuffle configuration. + +Record: + +```text +seed +test count +pass count +fail count +skip count +warnings +test duration +wall duration +``` + +The known E7 rate-limit flake may appear. + +If E7 appears: + +1. record the seed; +2. run that test in isolation; +3. run the complete suite once more with a new seed; +4. do not fix E7 in this branch; +5. ensure no storage failure is mislabeled as E7. + +A deterministic storage failure must be fixed before completion. + +--- + +# 29. Documentation updates + +Update: + +```text +docs/xii/architecture/inventory/storage-call-sites.md +``` + +For each call site, mark: + +```text +migrated_to_django_storage +legacy_signed_url_only +temporary_wrapper +not_storage_persistence +blocked +``` + +Update: + +```text +docs/xii/architecture/inventory/frontend-file-flow.md +``` + +Record: + +- base64 upload flow remaining; +- signed upload flow remaining; +- signed download flow remaining; +- which persistence operations now use Django Storage; +- exact scope required for IS-02. + +Update: + +```text +docs/xii/architecture/inventory/unresolved-items.md +``` + +only with real unresolved storage issues. + +Correct architecture documents so they explicitly state: + +```text +Django Storage API is the architecture. +MinIO through S3Storage is the default local profile. +Generic S3-compatible storage remains supported. +GCS is the initial GCP storage profile, not the only supported provider. +``` + +Do not rewrite unrelated task, cache, Redis or Terraform sections. + +--- + +# 30. Required implementation deliverables + +This phase should produce, as applicable: + +```text +django-storages dependency configuration +S3 and GCS backend dependencies +provider-selectable STORAGES configuration +patient storage alias +facility storage alias +report storage alias +preserved staticfiles alias +provider-neutral object-name helper +migrated ordinary storage CRUD +thin compatibility wrapper only where necessary +storage configuration tests +storage behavior tests +MinIO integration tests +GCS configuration tests +updated inventories +updated architecture wording +``` + +--- + +# 31. Prohibited work + +Do not: + +- implement multipart uploads; +- change frontend code; +- remove every signed URL endpoint if doing so requires the next transport phase; +- subclass storage backends merely to create signed uploads; +- introduce a custom GCS manager; +- introduce a custom S3 manager as a new primary abstraction; +- make GCP mandatory; +- require GCP credentials locally; +- remove MinIO; +- remove Celery; +- change Redis; +- fix locks; +- fix E7 rate limiting; +- add PostgreSQL cache; +- add PostgreSQL queues; +- add Cloud Tasks; +- add Cloud Run; +- add Cloud SQL infrastructure; +- add Terraform; +- change domain models for architectural purity; +- add repository patterns; +- reorganize CARE applications; +- push commits. + +--- + +# 32. Commit strategy + +Use focused commits. + +Recommended sequence: + +```text +chore(storage): add django-storages backends +feat(storage): configure portable storage aliases +refactor(storage): route file persistence through Django Storage +test(storage): cover aliases and MinIO integration +docs(storage): update modernization inventory +``` + +The exact sequence may differ if repository mechanics make another grouping clearer. + +Avoid a single giant commit. + +Do not push. + +--- + +# 33. Acceptance criteria + +IS-01 is complete only when all of the following are true: + +- `CARE_STORAGE_BACKEND` supports `s3` and `gcs`; +- `s3` is the default; +- local Docker uses MinIO through `S3Storage`; +- local startup requires no GCP values; +- GCS aliases can be configured through `GoogleCloudStorage`; +- patient, facility and report aliases are provider-neutral; +- static files remain on WhiteNoise; +- ordinary CARE storage persistence uses Django Storage; +- migrated modules no longer instantiate `boto3`; +- no manual GCS persistence implementation exists; +- provider switching requires settings changes only; +- MinIO integration tests pass; +- GCS configuration tests pass; +- the full CARE suite passes, subject only to the documented E7 flake procedure; +- remaining signed URL paths are precisely documented for IS-02; +- remaining base64 transport is precisely documented for IS-02; +- the branch contains no Redis, task, lock or GCP deployment changes. + +--- + +# 34. Final report + +At completion, report: + +1. current branch; +2. initial and final commit; +3. commits created; +4. dependencies added and resolved versions; +5. files created; +6. files modified; +7. storage-selection configuration; +8. aliases implemented; +9. local MinIO behavior; +10. generic S3-compatible behavior; +11. GCS configuration behavior; +12. ordinary legacy CRUD removed; +13. direct `boto3` persistence imports remaining and exact reasons; +14. compatibility wrapper remaining and exact callers; +15. signed upload paths remaining; +16. signed download paths remaining; +17. base64 upload path remaining; +18. whole-file memory reads remaining; +19. MinIO integration-test results; +20. GCS configuration-test results; +21. full test results with seeds and counts; +22. whether E7 occurred; +23. documentation updated; +24. unresolved storage issues; +25. exact recommended scope for IS-02 File Transport Modernization. + +Stop after IS-01. + +Do not begin IS-02. diff --git a/docs/xii/implementation/ES-02-file-transport-modernization.md b/docs/xii/implementation/ES-02-file-transport-modernization.md new file mode 100644 index 0000000000..a13f0bb566 --- /dev/null +++ b/docs/xii/implementation/ES-02-file-transport-modernization.md @@ -0,0 +1,1178 @@ +# ES-02: File Transport Modernization + +- **Status:** Draft +- **Related ADR:** ADR-0002: Server-Mediated File Transport +- **Depends on:** ADR-0001 and completed ES-01 +- **Target branch:** `feature/file-transport-modernization` + +--- + +# 1. Context + +ES-01 completed the storage modernization. + +The current storage architecture is now: + +```text +CARE + ↓ +Django Storage API + ↓ +django-storages + ├── S3Storage → MinIO / S3-compatible providers + └── GoogleCloudStorage → GCS +``` + +All object-storage persistence is provider-neutral. + +Signed upload and download URLs have been removed. + +Downloads now pass through CARE using Django and `FileResponse`. + +The remaining transport issue is the current upload contract. + +CARE still contains an upload flow in which file content is sent to Django as +base64-encoded data. + +That flow now persists correctly through Django Storage, but the HTTP transport +itself remains inefficient. + +The target of this phase is therefore narrowly defined: + +```text +base64-over-JSON upload + ↓ +multipart/form-data upload + ↓ +Django UploadedFile / upload handlers + ↓ +Django Storage API +``` + +This phase does not change the storage architecture established by ES-01. + +--- + +# 2. Repository State + +Before implementation, verify the current repository state. + +The active branch MUST be: + +```text +feature/file-transport-modernization +``` + +The branch MUST be based on the current `gcp` branch containing the merged and +completed ES-01 implementation. + +Before modifying code, report: + +```bash +git status +git branch --show-current +git log -5 --oneline +git remote -v +``` + +Verify: + +- the working tree is clean; +- `gcp` contains completed ADR-0001 / ES-01; +- `upstream` points to `https://github.com/ohcnetwork/care`; +- no unrelated local commits are present. + +Do not reset, rebase, merge unrelated work or push automatically. + +--- + +# 3. Required Documents + +Read these documents before implementation: + +```text +docs/xii/architecture/00-scope-and-goals.md +docs/xii/architecture/01-current-runtime.md +docs/xii/architecture/02-target-runtime.md +docs/xii/architecture/03-migration-plan.md +docs/xii/architecture/04-testing.md +docs/xii/architecture/07-configuration-reference.md + +docs/xii/adr/ADR-0001-django-storage.md +docs/xii/adr/ADR-0002-file-transport.md + +docs/xii/implementation/ES-01-storage.md + +docs/xii/architecture/inventory/storage-call-sites.md +docs/xii/architecture/inventory/frontend-file-flow.md +docs/xii/architecture/inventory/unresolved-items.md +``` + +If the repository uses different final paths, locate the actual committed files +and use those. + +The ADR defines the architectural decision. + +This ES defines the implementation requirements. + +The current source code is authoritative for implementation details. + +--- + +# 4. Objective + +Replace CARE's remaining base64-based file upload transport with normal HTTP +multipart upload handling. + +After this phase: + +- file uploads use `multipart/form-data`; +- Django receives files as `UploadedFile` objects; +- Django upload handlers determine memory vs temporary-file behavior; +- the complete uploaded file is not base64-encoded; +- the API does not require complete file content inside JSON; +- object persistence continues through Django Storage; +- downloads continue through CARE and Django Storage; +- no provider URL is returned to clients; +- no browser communicates directly with object storage; +- MinIO, S3-compatible storage and GCS all use the same HTTP API; +- upload validation and authorization remain centralized in CARE. + +This phase modernizes **HTTP file transport only**. + +--- + +# 5. Target Architecture + +The upload path SHALL become: + +```text +Client + | + | multipart/form-data + v +CARE Django API + | + | request.FILES / UploadedFile + v +CARE validation and authorization + | + v +Django Storage API + | + v +Configured storage backend +``` + +The download path remains: + +```text +Client + | + | authenticated CARE request + v +CARE Django API + | + | authorization + v +Django Storage API + | + | file-like object + v +FileResponse +``` + +The HTTP contract SHALL remain provider-independent. + +--- + +# 6. Scope + +This phase includes: + +- replacing base64 file payloads; +- multipart request parsing; +- Django `UploadedFile` handling; +- upload size configuration; +- validation of upload metadata; +- preserving existing authorization semantics; +- preserving existing object naming; +- preserving provider-neutral storage aliases; +- updating API schemas; +- updating backend tests; +- updating frontend-facing API contracts in this repository where applicable; +- removal of obsolete base64 transport code; +- removal of obsolete upload DTO/schema fields; +- documentation updates. + +This phase does not include: + +- Cloud Tasks; +- Celery migration; +- retry redesign; +- Redis; +- cache; +- locks; +- rate-limit fixes; +- Cloud Run; +- Cloud SQL; +- Terraform; +- CI/CD; +- provider-native multipart upload; +- signed URLs; +- CDN design; +- resumable uploads; +- antivirus scanning; +- large-video ingestion architecture. + +--- + +# 7. Out of Scope + +Do not: + +- reintroduce signed URLs; +- expose bucket URLs; +- expose storage-provider names; +- implement browser-to-bucket uploads; +- redesign Django Storage; +- create a new upload abstraction layer; +- modify unrelated models; +- modify task execution; +- modify Redis configuration; +- fix E7; +- change report-generation retry behavior; +- add cloud deployment resources. + +Remain strictly inside file transport. + +--- + +# 8. Design Rules + +## 8.1 Django owns HTTP upload parsing + +Use Django and Django REST Framework's normal multipart support. + +Do not manually parse multipart bodies. + +Do not decode the complete upload manually. + +Use: + +```text +request.FILES +UploadedFile +TemporaryUploadedFile +InMemoryUploadedFile +``` + +or the repository's existing DRF equivalents. + +## 8.2 Django Storage remains the persistence boundary + +After receiving and validating the `UploadedFile`, persistence SHALL continue +through the provider-neutral storage seam created by ES-01. + +No transport code may instantiate: + +- boto3; +- GCS clients; +- MinIO clients; +- provider SDKs. + +## 8.3 No base64 file transport + +The target upload API SHALL not accept file content encoded as base64. + +Any current field containing complete file content as: + +```text +base64 +data URL +encoded JSON string +``` + +shall be removed from the final production upload contract. + +## 8.4 No provider-specific transport behavior + +Upload behavior SHALL be identical for: + +```text +MinIO +AWS S3 +generic S3-compatible storage +Google Cloud Storage +``` + +Provider switching must remain configuration-only. + +--- + +# 9. Current Upload Flow Inventory + +Before changing code, inspect the exact current base64 upload flow identified by +the ES-01 inventory. + +Document: + +- endpoint path; +- viewset or API function; +- serializer or request schema; +- base64 field name; +- metadata fields; +- authorization path; +- object-name generation; +- storage alias selection; +- model updates; +- response schema; +- frontend or tests that call it. + +Verify whether more than one base64 upload path exists. + +Do not assume there is only one because the inventory previously identified one. + +Update the inventory if the source has changed. + +--- + +# 10. Multipart API Contract + +The upload endpoint SHALL accept: + +```text +Content-Type: multipart/form-data +``` + +The request SHALL contain one uploaded file field. + +Preferred field name: + +```text +file +``` + +If the current CARE API already uses another established field name and changing +it would create unnecessary compatibility work, preserve the established name. + +Metadata SHOULD remain ordinary multipart fields. + +Examples may include: + +```text +file_type +name +patient +facility +encounter +metadata +``` + +Only verified current fields SHALL be preserved. + +Do not create speculative metadata fields. + +--- + +# 11. Serializer / Request Validation + +Use DRF's file-aware serializer fields where appropriate. + +Prefer: + +```python +serializers.FileField() +``` + +or equivalent project-native request schema support. + +Validation SHALL occur before persistence when possible. + +Validate: + +- file presence; +- maximum size; +- extension; +- allowed MIME type; +- logical file type; +- related object identifiers; +- authorization; +- filename safety. + +Do not trust browser-provided MIME type as the only source of truth when CARE +already has stronger validation logic. + +Preserve existing validation behavior unless the base64 implementation itself +prevented correct validation. + +--- + +# 12. File Size Limits + +Explicit file-size limits SHALL be defined. + +The implementation SHALL inspect current CARE limits before introducing new +ones. + +The relevant settings may include: + +```text +CARE_MAX_UPLOAD_SIZE +FILE_UPLOAD_MAX_MEMORY_SIZE +DATA_UPLOAD_MAX_MEMORY_SIZE +``` + +The actual names must match the configuration reference. + +The implementation SHALL ensure: + +- small supported files may remain in memory; +- larger supported files use Django temporary-file handling; +- unsupported oversized files are rejected cleanly; +- the application does not base64-decode large payloads into memory; +- Cloud/runtime-specific limits are not hardcoded in domain code. + +Do not invent an arbitrary maximum if CARE already defines one. + +If no maximum exists, add an explicit conservative setting and document it. + +--- + +# 13. Temporary Files + +Django may store larger uploads in temporary files. + +Temporary files SHALL: + +- use Django's normal upload-handler lifecycle; +- remain ephemeral; +- not become durable storage; +- be closed after use; +- not be manually copied unless required. + +Cloud or local runtime filesystem behavior SHALL not leak into application +semantics. + +The implementation MAY configure: + +```text +FILE_UPLOAD_TEMP_DIR=/tmp +``` + +only if necessary. + +Do not assume `/tmp` is durable. + +--- + +# 14. Storage Save Path + +The upload implementation SHALL pass the Django uploaded-file object or +file-like object directly to the relevant storage implementation where possible. + +Preferred conceptual behavior: + +```python +uploaded_file = serializer.validated_data["file"] + +stored_name = storage.save( + object_name, + uploaded_file, +) +``` + +Do not do: + +```python +content = uploaded_file.read() +storage.save(name, ContentFile(content)) +``` + +unless a verified downstream requirement forces a complete read. + +Avoid unnecessary memory copies. + +--- + +# 15. Object Naming + +Object naming SHALL remain exactly compatible with ES-01. + +The existing provider-neutral object-name helper remains authoritative. + +The HTTP upload phase SHALL not create a second naming function. + +Object names remain storage-relative and provider-neutral. + +Do not include: + +- bucket names; +- endpoints; +- URLs; +- provider prefixes. + +--- + +# 16. Logical Storage Alias Selection + +Upload transport SHALL continue selecting the appropriate logical alias: + +```text +patient +facility +report +``` + +The transport layer SHALL not inspect `CARE_STORAGE_BACKEND`. + +Alias selection is based on CARE domain intent, not infrastructure. + +Tests must prove that multipart transport does not alter alias selection. + +--- + +# 17. Authorization + +Preserve CARE's existing upload authorization model. + +The new multipart endpoint SHALL not weaken: + +- patient access; +- facility access; +- encounter access; +- ownership checks; +- role checks; +- organization boundaries. + +The transport modernization SHALL not redesign authorization logic. + +Existing authorization tests SHALL continue to pass. + +Add focused tests if current base64 tests did not verify authorization +adequately. + +--- + +# 18. Database and Storage Consistency + +Object storage and PostgreSQL do not share an atomic transaction. + +The implementation SHALL preserve or improve the current failure behavior. + +Analyze the current sequence: + +```text +validate +save object +write DB +``` + +or: + +```text +write DB +save object +``` + +Document the actual current behavior. + +Handle partial failures explicitly. + +At minimum: + +- storage failure must not report successful upload; +- database failure after storage save must not leave an undetectable completed + record; +- existing cleanup behavior must remain compatible; +- incomplete-object cleanup logic must still work. + +Do not introduce a complex distributed transaction framework. + +Use the smallest correct compensation behavior. + +--- + +# 19. Response Contract + +The successful upload response SHALL remain provider-neutral. + +It MAY include: + +- CARE object identifier; +- metadata; +- relative CARE `download_url`; +- filename; +- MIME type; +- size; +- status. + +It SHALL NOT include: + +- bucket URL; +- S3 URL; +- GCS URL; +- signed URL; +- storage credentials; +- provider-specific object metadata unless CARE already exposes it for a valid + reason. + +If the previous base64 endpoint returned provider details, remove them. + +--- + +# 20. Downloads + +ES-01 already implemented server-mediated downloads. + +Preserve them. + +Do not re-architect download transport unless a defect is discovered. + +Verify that upload-generated objects can be downloaded through: + +```text +/files/{id}/download/ +/template_reports/{id}/download/ +/assets/facility/{id}/cover_image/ +/assets/user/{username}/profile_picture/ +``` + +or the actual committed equivalents. + +The new multipart upload must integrate cleanly with those routes. + +--- + +# 21. Content-Disposition + +Preserve existing inline vs attachment behavior. + +Do not broaden `SAFE_INLINE_FORMATS` without a verified requirement. + +Filename handling must remain safe against header injection. + +Tests should cover representative inline and attachment formats. + +--- + +# 22. MIME Types + +Preserve existing CARE MIME validation where possible. + +The implementation SHALL distinguish: + +```text +request-declared MIME +validated MIME +response MIME +``` + +The browser-declared MIME type SHALL not automatically be treated as trusted. + +If the existing base64 endpoint used extension-only validation, preserve +behavior first unless the ADR or current security policy explicitly requires +stronger validation. + +Do not introduce heavyweight file-inspection dependencies in this phase without +a demonstrated requirement. + +--- + +# 23. Base64 Removal + +Once the multipart path is verified, remove: + +- base64 decoding logic; +- base64 request fields; +- base64-specific validators; +- data-URL parsing code; +- documentation describing base64 upload; +- tests whose sole purpose is preserving the old transport. + +Do not retain a hidden base64 fallback. + +This is a greenfield deployment. + +There is no production client that requires compatibility. + +--- + +# 24. API Schema + +Update the API/OpenAPI schema so the upload request is represented as a file +upload. + +The generated schema SHOULD show: + +```text +multipart/form-data +``` + +and the correct binary file field. + +Remove base64-body documentation. + +Verify the schema generator does not emit the old JSON contract. + +--- + +# 25. Frontend Contract + +If frontend code is included in this repository and directly consumes the +upload endpoint, update it. + +The client SHALL: + +- send `FormData`; +- append the file object; +- include required metadata fields; +- call the CARE endpoint; +- consume CARE's provider-neutral response; +- never contact object storage. + +If the frontend is outside this repository, document the exact API contract +change instead of inventing or modifying external code. + +Do not restore compatibility solely to avoid frontend coordination. + +--- + +# 26. Cover Images and Profile Pictures + +ES-01 already moved cover-image and profile-picture downloads behind CARE. + +Inspect whether their upload path also uses the base64 endpoint. + +If yes, migrate it to multipart through the same provider-neutral API or the +smallest existing specialized endpoint. + +Do not create duplicated upload transport logic. + +If those assets already use ordinary multipart handling, leave them unchanged +and document that fact. + +--- + +# 27. Report Files + +Do not modify report-generation retry policy in this phase. + +The existing storage-adjacent: + +```text +botocore.ClientError +``` + +in `report_generation.py` remains assigned to ES-03. + +This phase may verify that generated reports remain downloadable after upload +transport changes, but SHALL not change Celery retry behavior. + +--- + +# 28. E7 + +The documented E7 parallel-test defect remains out of scope. + +Do not fix: + +- rate-limit cache keys; +- Redis cache clearing; +- test worker prefixes; +- favorites cache behavior. + +When interpreting parallel test failures, follow the existing E7 procedure. + +A deterministic transport or storage failure must not be classified as E7. + +--- + +# 29. Allowed Modifications + +Claude MAY modify: + +- upload viewsets; +- upload serializers; +- upload request schemas; +- upload helpers; +- upload tests; +- API schema tests; +- provider-neutral file transport helpers; +- settings related to upload size; +- frontend code in this repository if it calls the changed endpoint; +- file-flow inventory; +- ADR-0002 implementation checklist; +- ES-02 documentation; +- configuration reference where actual upload settings change. + +Claude MAY remove: + +- base64 transport code; +- base64-specific tests; +- base64-specific schema fields; +- obsolete upload compatibility code. + +--- + +# 30. Forbidden Modifications + +Claude SHALL NOT modify: + +- Redis configuration; +- rate limiting; +- distributed locks; +- Celery dispatch architecture; +- Cloud Tasks; +- Cloud Run; +- Cloud SQL; +- Terraform; +- CI/CD; +- unrelated domain models; +- Django ORM architecture; +- report retry behavior; +- SMS boto3 integration; +- storage provider architecture established in ES-01. + +No migrations should be required unless the current base64 representation is +stored in the database, which must be verified before any migration is added. + +Do not add a migration merely because an API serializer changed. + +--- + +# 31. Tests — Focused + +Add focused multipart tests. + +At minimum: + +## Successful upload + +- valid authenticated request; +- authorized caller; +- multipart file; +- correct logical alias; +- correct object name; +- storage save succeeds; +- DB record is correct; +- response is provider-neutral; +- download works. + +## Missing file + +Reject multipart requests without the file field. + +## Oversized file + +Reject files above the configured maximum. + +## Extension validation + +Test: + +- allowed; +- blocked; +- uppercase; +- double extension where relevant. + +## MIME validation + +Test: + +- accepted MIME; +- mismatched MIME; +- missing MIME; +- unsafe MIME. + +## Authorization + +Test unauthorized upload attempts against the appropriate CARE scope. + +## Temporary-file path + +Add at least one test with a file larger than: + +```text +FILE_UPLOAD_MAX_MEMORY_SIZE +``` + +using a test-specific low threshold where necessary. + +Verify the handler receives a temporary-file-backed upload where the framework +supports deterministic testing. + +Do not depend on huge test fixtures. + +## Storage failures + +Simulate: + +- save failure; +- DB failure after save where applicable. + +Verify consistent cleanup or explicit incomplete-state handling. + +## Response + +Assert no response contains: + +```text +signed_url +read_signed_url +bucket +endpoint +provider URL +``` + +unless `bucket` is a legitimate unrelated domain field. + +--- + +# 32. MinIO Integration + +Run the multipart upload flow against real local MinIO through `S3Storage`. + +Verify: + +```text +multipart request +→ Django +→ S3Storage +→ MinIO +→ download through Django +``` + +Test at least one complete round trip. + +MinIO remains the mandatory local integration profile. + +--- + +# 33. GCS Configuration / Provider-Neutral Tests + +Ordinary local tests must not require live GCS. + +Use configuration or test storage to prove: + +- multipart code does not branch on provider; +- alias selection remains identical under `gcs`; +- transport code receives Django Storage objects; +- no S3 assumptions remain. + +If an optional live GCS test already exists from ES-01, extend it only if +credentials are available. + +Do not make live GCS mandatory for the normal test suite. + +--- + +# 34. API Schema Tests + +Verify the generated API schema for the upload endpoint declares: + +```text +multipart/form-data +``` + +and a binary/file field. + +Assert the old base64 field is absent. + +If CARE uses typed OpenAPI request/response specs, update them consistently. + +--- + +# 35. Regression Tests + +After focused tests pass: + +1. rebuild the image if dependencies or runtime settings changed; +2. start the official local stack; +3. verify all services healthy; +4. run migrations; +5. verify permission and value-set synchronization; +6. run fixtures as appropriate; +7. run the full serial suite; +8. run the documented parallel suite. + +Record: + +```text +seed +test count +pass count +failure count +skip count +test duration +wall duration +``` + +The serial full suite must be green. + +Parallel failures may be accepted only when they are demonstrated to belong to +the already documented E7 defect class. + +No transport-related deterministic failure is acceptable. + +--- + +# 36. Performance Sanity Check + +This phase does not require a benchmark suite, but it SHALL verify that the new +multipart path avoids base64 expansion. + +For one representative test file, record or verify qualitatively: + +```text +old transport: +JSON + base64 + +new transport: +multipart binary +``` + +Do not introduce a performance framework. + +The implementation should not materialize duplicate full-size byte buffers. + +--- + +# 37. Documentation Updates + +Update: + +```text +docs/xii/architecture/inventory/frontend-file-flow.md +``` + +to mark: + +```text +base64 upload -> removed +multipart upload -> implemented +server-mediated download -> implemented +signed upload -> removed +signed download -> removed +``` + +Update: + +```text +docs/xii/architecture/inventory/storage-call-sites.md +``` + +only if transport changes affect documented call sites. + +Update: + +```text +docs/xii/adr/ADR-0002-file-transport.md +``` + +implementation checklist. + +Update the configuration reference for final upload-size settings. + +Update API documentation where the request contract changes. + +Do not modify unrelated ADRs. + +--- + +# 38. Commit Strategy + +Use small logical commits. + +Recommended sequence: + +```text +refactor(files): replace base64 upload with multipart transport + +test(files): cover multipart upload and provider-neutral round trips + +docs(files): complete file transport modernization +``` + +If API schema and frontend changes are substantial, they may be separate +focused commits. + +Do not create a giant mixed commit. + +Do not squash. + +Do not push. + +--- + +# 39. Acceptance Criteria + +ES-02 is complete only when all of the following are true: + +- upload uses `multipart/form-data`; +- Django receives an `UploadedFile`; +- complete file content is no longer base64-encoded; +- base64 upload fields and decoder code are removed; +- uploads pass through CARE; +- downloads pass through CARE; +- Django Storage remains the only persistence boundary; +- no provider URL is returned; +- no browser-to-bucket transport exists; +- MinIO round-trip upload/download passes; +- GCS profile requires no transport-specific code; +- provider switching remains configuration-only; +- upload size limits are explicit; +- larger supported uploads use Django temporary-file behavior where applicable; +- authorization remains intact; +- file validation remains intact; +- API schema reflects multipart upload; +- frontend contract is updated or precisely documented; +- serial full regression is green; +- no deterministic file-transport regression remains; +- ADR-0002 implementation checklist is updated. + +--- + +# 40. Final Report + +At completion provide: + +1. branch; +2. initial and final commit; +3. commits created; +4. files created; +5. files modified; +6. files deleted; +7. old base64 endpoint behavior; +8. final multipart endpoint contract; +9. serializer/request-schema changes; +10. upload size configuration; +11. temporary-file behavior; +12. authorization behavior; +13. validation behavior; +14. database/storage consistency handling; +15. MinIO round-trip result; +16. GCS/provider-neutral test result; +17. API schema result; +18. frontend changes or documented external contract; +19. focused test counts; +20. serial full-suite result; +21. parallel result and any E7 occurrences; +22. documentation updated; +23. unresolved file-transport items; +24. deviations from ADR-0002 or ES-02; +25. final verdict: + +```text +READY TO MERGE +``` + +or: + +```text +NOT READY TO MERGE +``` + +Stop after ES-02. + +Do not begin ES-03. diff --git a/docs/xii/prompts/00-Complete-CARE-Runtime-Inventory b/docs/xii/prompts/00-Complete-CARE-Runtime-Inventory new file mode 100644 index 0000000000..63d5b7f900 --- /dev/null +++ b/docs/xii/prompts/00-Complete-CARE-Runtime-Inventory @@ -0,0 +1,458 @@ +# Claude Code Prompt — Phase 0: Complete CARE Runtime Inventory + +You are working inside a fork of: + +```text +https://github.com/ohcnetwork/care +``` + +The official upstream development branch is: + +```text +upstream/develop +``` + +The maintained integration branch is: + +```text +gcp +``` + +The current working branch should be: + +```text +feature/gcp-phase-0-inventory +``` + +## Objective + +Complete Phase 0 of the greenfield GCP implementation plan. + +Do not implement GCP support yet. + +Do not modify application behavior. + +Do not refactor CARE. + +Do not add abstractions. + +Your task is only to inspect the current repository, verify the existing GCP documentation against the real code and create a complete call-site inventory. + +Read these documents first: + +```text +docs/xii/architecture/00-scope-and-goals.md +docs/xii/architecture/01-current-runtime.md +docs/xii/architecture/02-target-runtime.md +docs/xii/architecture/03-migration-plan.md +docs/xii/architecture/04-testing.md +docs/xii/architecture/05-upstream-sync.md +docs/xii/architecture/06-operations.md +docs/xii/architecture/07-configuration-reference.md +``` + +The architectural goal is narrowly defined: + +* Deploy CARE greenfield on GCP. +* Use Cloud Run. +* Use Cloud SQL. +* Use Django Storage API and `django-storages`. +* All uploads and downloads pass through Django. +* Do not use direct browser-to-bucket uploads. +* Use Cloud Tasks by default. +* Support PostgreSQL for cache and appropriate shared state. +* Keep Redis optional. +* Preserve local Redis, Celery and MinIO compatibility. +* Do not redesign CARE or abstract Django ORM. + +## Preliminary verification + +Before inspecting code, report: + +```bash +git status +git branch --show-current +git remote -v +git log -1 --oneline +``` + +Verify that: + +* the working tree is clean; +* the branch is derived from `gcp`; +* `origin` exists; +* `upstream` points to the official CARE repository. + +Do not reset, merge, rebase, push or modify branches automatically. + +## Required inventory + +Create: + +```text +docs/xii/architecture/inventory/ +├── storage-call-sites.md +├── task-call-sites.md +├── cache-and-redis.md +├── frontend-file-flow.md +├── runtime-and-deployment.md +├── plugin-impact.md +└── unresolved-items.md +``` + +### Storage inventory + +Locate every use of: + +```text +files_manager +S3FilesManager +boto3 +botocore +signed_url +read_signed_url +put_object +get_object +delete_object +delete_objects +BUCKET_PROVIDER +BUCKET_ENDPOINT +BUCKET_EXTERNAL_ENDPOINT +FILE_UPLOAD_BUCKET +FACILITY_S3_BUCKET +``` + +For each call site, document: + +* file path; +* symbol or function; +* line number; +* caller; +* operation; +* logical bucket; +* whether the call is used by API, task, command or plugin; +* whether it exposes a direct object-storage URL; +* whether it reads the complete file into memory; +* whether it can be replaced by Django Storage API; +* uncertainty requiring further analysis. + +Do not modify storage code. + +### Task inventory + +Locate every use of: + +```text +@shared_task +@app.task +.delay( +.apply_async( +send_task( +AsyncResult +CELERY_RESULT_BACKEND +add_periodic_task +crontab +``` + +For every task, document: + +* task name; +* source file; +* call sites; +* payload; +* return value; +* whether callers use the task ID; +* whether callers read the result; +* retry policy; +* expiry; +* expected duration when inferable; +* database effects; +* storage effects; +* email or external-service effects; +* idempotency risks; +* periodic schedule; +* plugin ownership; +* likely target classification. + +Classifications are limited to: + +```text +cloud_tasks_candidate +cloud_run_job_candidate +synchronous_candidate +celery_compatibility +requires_analysis +``` + +Do not migrate tasks. + +### Cache and Redis inventory + +Locate every use of: + +```text +REDIS_URL +django_redis +django.core.cache +from django.core.cache import cache +cache. +redis +Redis +django_ratelimit +CELERY_BROKER_URL +CELERY_RESULT_BACKEND +``` + +Classify every use as: + +```text +celery_broker +celery_result_backend +performance_cache +shared_cache +report_progress +rate_limit +distributed_lock +transient_state +session +health_check +direct_redis +unknown +``` + +For each use, document whether it could use: + +```text +PostgreSQL database cache +explicit PostgreSQL model +LocMem +Redis-compatible backend +not applicable +``` + +Do not claim PostgreSQL is suitable without examining the required semantics. + +Do not modify cache or Redis code. + +### Frontend file-flow inventory + +Inspect the repository for frontend-facing contracts related to files. + +Document: + +* upload initiation endpoint; +* upload completion endpoint; +* signed URL response; +* download endpoint; +* direct bucket URL response; +* serializers; +* OpenAPI schema; +* tests; +* frontend repository references, if present; +* exact API changes required to make all traffic pass through Django. + +Do not implement the new file API. + +### Runtime and deployment inventory + +Inspect: + +```text +Dockerfiles +docker-compose files +Makefile +startup scripts +health checks +GitHub Actions +deployment manifests +dependency files +lockfiles +settings modules +``` + +Document: + +* current API command; +* current worker command; +* migration behavior; +* static collection; +* Redis waits; +* database waits; +* Celery Beat startup; +* production image availability; +* CI commands; +* dependency manager; +* relevant Python and Django versions; +* existing GCP-related code, if any. + +### Plugin impact + +Determine how CARE loads plugins. + +Identify any installed or bundled plugins that may introduce: + +* Celery tasks; +* Redis dependencies; +* direct S3 or boto3 use; +* signed URLs; +* custom health checks; +* custom startup behavior; +* additional migrations. + +Do not assume external plugins are compatible. + +Document what can and cannot be determined from the repository. + +## Documentation verification + +Compare the real repository against: + +```text +docs/xii/architecture/01-current-runtime.md +``` + +Correct factual inaccuracies only. + +`01-current-runtime.md` must remain purely descriptive. + +Do not insert target architecture, recommendations or migration instructions +into that document. + +If the repository has changed since the document was written, update: + +```text +reviewed +source commit +affected sections +``` + +Do not rewrite the other GCP documents unless a verified code fact directly +contradicts them. + +List any such contradictions in: + +```text +docs/xii/architecture/inventory/unresolved-items.md +``` + +## Baseline commands + +Discover the official commands from the repository before running them. + +Run the safest available equivalents of: + +```bash +make build +make up +make load-fixtures +make test +``` + +Do not invent commands. + +Do not delete volumes. + +Do not use destructive database reset commands. + +Record: + +* exact command; +* success or failure; +* duration if readily available; +* test count; +* failures; +* skipped tests; +* unhealthy services; +* relevant logs. + +Store the baseline in: + +```text +docs/xii/architecture/inventory/runtime-and-deployment.md +``` + +If an external dependency prevents execution, document the exact blocker. + +## Prohibited work + +Do not: + +* add `django-storages`; +* add GCP dependencies; +* add Terraform; +* modify settings behavior; +* migrate storage; +* create upload endpoints; +* create Cloud Tasks handlers; +* make Redis optional; +* introduce PostgreSQL cache; +* change models; +* add migrations; +* modify the frontend; +* commit generated secrets; +* redesign the project; +* add repository patterns; +* create generic ports or adapters; +* push changes. + +## Quality requirements + +Every inventory statement must reference: + +* an exact file; +* an exact symbol; +* and, where practical, a line number. + +Clearly distinguish: + +```text +verified +inferred +unknown +``` + +Do not present inference as fact. + +Avoid generic architecture advice. + +Focus on the repository as it exists. + +## Tests + +Because this phase changes documentation only: + +* run Markdown formatting or linting if the repository provides it; +* rerun no application tests solely because documentation changed, except for + the initial baseline; +* verify all documented file paths exist; +* verify all documented symbols exist. + +## Commit + +After completing the inventory, create one commit: + +```text +docs(gcp): complete runtime call-site inventory +``` + +Do not push. + +## Final report + +Provide: + +1. current branch and commit; +2. files created; +3. files modified; +4. factual corrections made to `01-current-runtime.md`; +5. task count; +6. storage call-site count; +7. cache and Redis call-site count; +8. plugin uncertainties; +9. baseline command results; +10. blockers; +11. recommended next implementation phase. + +Stop after Phase 0. + +Do not begin implementation.