diff --git a/.env.example b/.env.example index 9c2c961..9836e6e 100644 --- a/.env.example +++ b/.env.example @@ -1,5 +1,9 @@ # Copy to .env and adjust for your environment: # cp .env.example .env +# +# Full variable reference (authoritative): +# https://ndevu12.github.io/Research_Assistant_Model/configuration/environment-variables/ +# or docs/configuration/environment-variables.md in the repo # ============================================================================= # Retrieval APIs (optional) @@ -25,6 +29,7 @@ RA_LLM__PROVIDER=ollama RA_LLM__MODEL=auto # RA_LLM__MODEL=llama3.1:8b +# With or without /v1 — Ollama providers normalize via normalize_openai_base_url() RA_LLM__BASE_URL=http://localhost:11434/v1 RA_LLM__API_KEY=ollama # OLLAMA_API_KEY=ollama diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml new file mode 100644 index 0000000..cbabbca --- /dev/null +++ b/.github/workflows/docs.yml @@ -0,0 +1,87 @@ +name: Deploy docs + +on: + push: + branches: [main] + paths: + - 'docs/**' + - 'mkdocs.yml' + - 'contributing.md' + - 'scripts/check_docs_policy.py' + - '.github/workflows/docs.yml' + pull_request: + paths: + - 'docs/**' + - 'mkdocs.yml' + - 'contributing.md' + - 'scripts/check_docs_policy.py' + - '.github/workflows/docs.yml' + workflow_dispatch: + +permissions: + contents: read + pages: write + id-token: write + +concurrency: + group: pages + cancel-in-progress: false + +jobs: + build: + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: '3.13' + + - name: Install docs tooling + run: pip install mkdocs mkdocs-material mkdocs-mermaid2-plugin linkchecker + + - name: Docs policy (scaffold rejection) + run: python scripts/check_docs_policy.py + + - name: Build site + env: + NO_MKDOCS_2_WARNING: 1 + run: mkdocs build --strict + + - name: Check internal links + run: | + linkchecker \ + --no-warnings \ + --ignore-url="^https://github.com/" \ + --ignore-url="^https://pypi.org/" \ + --ignore-url="^https://pipenv.pypa.io/" \ + --ignore-url="^https://squidfunk.github.io/" \ + --ignore-url="^https://unpkg.com/" \ + --ignore-url="^https://api\." \ + --ignore-url="^https://doi.org/" \ + --ignore-url="^https://www\.crossref\.org/" \ + --ignore-url="^https://openalex.org/" \ + --ignore-url="^https://www\.semanticscholar\.org/" \ + --ignore-url="^https://arxiv\.org/" \ + --ignore-url="^https://ollama\.com/" \ + site/index.html + + - name: Upload Pages artifact + if: github.event_name == 'push' && github.ref == 'refs/heads/main' + uses: actions/upload-pages-artifact@v3 + with: + path: site + + deploy: + needs: build + if: github.event_name == 'push' && github.ref == 'refs/heads/main' + runs-on: ubuntu-latest + environment: + name: github-pages + url: ${{ steps.deployment.outputs.page_url }} + steps: + - name: Deploy to GitHub Pages + id: deployment + uses: actions/deploy-pages@v4 diff --git a/.gitignore b/.gitignore index f07d9f2..64000e1 100644 --- a/.gitignore +++ b/.gitignore @@ -61,3 +61,7 @@ reports/ # Python cache __pycache__/ pytest_cache/ + +# MkDocs build output +site/ +.cache/ diff --git a/Pipfile b/Pipfile index cfd44e6..8ff707a 100644 --- a/Pipfile +++ b/Pipfile @@ -24,6 +24,10 @@ torch = {version = "*", index = "pytorch-cpu"} rich = "*" [dev-packages] +mkdocs = "*" +mkdocs-material = "*" +mkdocs-mermaid2-plugin = "*" +linkchecker = "*" [requires] python_version = "3.13" diff --git a/Pipfile.lock b/Pipfile.lock index 874abcb..9d169ce 100644 --- a/Pipfile.lock +++ b/Pipfile.lock @@ -1,7 +1,7 @@ { "_meta": { "hash": { - "sha256": "c37ffadcf29936b88f736281e831c33d8d31187b04d86c3ddda2de52512983d1" + "sha256": "368c72cfc2325fe573783ad05d073cd094a11284f8013fe340b3c2338567b0dd" }, "pipfile-spec": 6, "requires": { @@ -548,11 +548,11 @@ }, "cohere": { "hashes": [ - "sha256:e5ade4423b928b01ff2038980e1b62b2a5bb412c8ab83e30882753b810a5509f", - "sha256:f15592ec60d8cf12f01563db94ec28c388c61269d9617f23c2d6d910e505344e" + "sha256:88ca34c91e634a64227f3eaa599d1af364b9d7bbc9336af93473fdced355617d", + "sha256:de565f04f805b909bd1040b70fad916dc7699c977c2f532289097bf3491d6cc3" ], - "markers": "python_version >= '3.9' and python_version < '4.0'", - "version": "==5.21.1" + "markers": "python_version >= '3.10' and python_version < '4.0'", + "version": "==7.0.0" }, "cryptography": { "hashes": [ @@ -3573,5 +3573,629 @@ "version": "==4.1.0" } }, - "develop": {} + "develop": { + "babel": { + "hashes": [ + "sha256:b80b99a14bd085fcacfa15c9165f651fbb3406e66cc603abf11c5750937c992d", + "sha256:e2b422b277c2b9a9630c1d7903c2a00d0830c409c59ac8cae9081c92f1aeba35" + ], + "markers": "python_version >= '3.8'", + "version": "==2.18.0" + }, + "backrefs": { + "hashes": [ + "sha256:4989bb9e1e99eb23647c7160ed51fb21d0b41b5d200f2d3017da41e023097e82", + "sha256:a0fa7360c63509e9e077e174ef4e6d3c21c8db94189b9d957289ae6d794b9475", + "sha256:a6448b28180e3ca01134c9cf09dcebafad8531072e09903c5451748a05f24bc9", + "sha256:b57cd227ea556b0aed3dc9b8da4628db4eabc0402c6d7fcfc69283a93955f7e9", + "sha256:ca42ce6a49ace3d75684dfa9937f3373902a63284ecb385ce36d15e5dcb41c12", + "sha256:f2c52955d631b9e1ac4cd56209f0a3a946d592b98e7790e77699339ae01c102a" + ], + "markers": "python_version >= '3.10'", + "version": "==7.0" + }, + "beautifulsoup4": { + "hashes": [ + "sha256:0918bfe44902e6ad8d57732ba310582e98da931428d231a5ecb9e7c703a735bb", + "sha256:6292b1c5186d356bba669ef9f7f051757099565ad9ada5dd630bd9de5fa7fb86" + ], + "markers": "python_full_version >= '3.7.0'", + "version": "==4.14.3" + }, + "certifi": { + "hashes": [ + "sha256:3c52e209ba0a4ad7aebe60436a4ab349c39e1e602e8c134221e546902ad25897", + "sha256:69dea482ab64caa7b9f6aba1c6bf48bb6a5448d1c0f1b17ab42ad8c763a5344d" + ], + "markers": "python_version >= '3.7'", + "version": "==2026.5.20" + }, + "charset-normalizer": { + "hashes": [ + "sha256:007d05ec7321d12a40227aae9e2bc6dca73f3cb21058999a1df9e193555a9dcc", + "sha256:03853ed82eeebbce3c2abfdbc98c96dc205f32a79627688ac9a27370ea61a49c", + "sha256:07d9e39b01743c3717745f4c530a6349eadbfa043c7577eef86c502c15df2c67", + "sha256:08e721811161356f97b4059a9ba7bafb23ea5ee2255402c42881c214e173c6b4", + "sha256:0c96c3b819b5c3e9e165495db84d41914d6894d55181d2d108cc1a69bfc9cce0", + "sha256:0ea948db76d31190bf08bd371623927ee1339d5f2a0b4b1b4a4439a65298703c", + "sha256:0f7eb884681e3938906ed0434f20c63046eacd0111c4ba96f27b76084cd679f5", + "sha256:12a6fff75f6bc66711b73a2f0addfc4c8c15a20e805146a02d147a318962c444", + "sha256:12d8baf840cc7889b37c7c770f478adea7adce3dcb3944d02ec87508e2dcf153", + "sha256:14265bfe1f09498b9d8ec91e9ec9fa52775edf90fcbde092b25f4a33d444fea9", + "sha256:16d971e29578a5e97d7117866d15889a4a07befe0e87e703ed63cd90cb348c01", + "sha256:177a0ba5f0211d488e295aaf82707237e331c24788d8d76c96c5a41594723217", + "sha256:1a87ca9d5df6fe460483d9a5bbf2b18f620cbed41b432e2bddb686228282d10b", + "sha256:1c2a768fdd44ee4a9339a9b0b130049139b8ce3c01d2ce09f67f5a68048d477c", + "sha256:1c2aed2e5e41f24ea8ef1590b8e848a79b56f3a5564a65ceec43c9d692dc7d8a", + "sha256:1dc8b0ea451d6e69735094606991f32867807881400f808a106ee1d963c46a83", + "sha256:1efde3cae86c8c273f1eb3b287be7d8499420cf2fe7585c41d370d3e790054a5", + "sha256:202389074300232baeb53ae2569a60901f7efadd4245cf3a3bf0617d60b439d7", + "sha256:203104ed3e428044fd943bc4bf45fa73c0730391f9621e37fe39ecf477b128cb", + "sha256:2257141f39fe65a3fdf38aeccae4b953e5f3b3324f4ff0daf9f15b8518666a2c", + "sha256:298930cec56029e05497a76988377cbd7457ba864beeea92ad7e844fe74cd1f1", + "sha256:2cd4a60d0e2fb04537162c62bbbb4182f53541fe0ede35cdf270a1c1e723cc42", + "sha256:2d6eb928e13016cea4f1f21d1e10c1cebd5a421bc57ddf5b1142ae3f86824fab", + "sha256:2fe249cb4651fd12605b7288b24751d8bfd46d35f12a20b1ba33dea122e690df", + "sha256:30b8d1d8c52a48c2c5690e152c169b673487a2a58de1ec7393196753063fcd5e", + "sha256:320ade88cfb846b8cd6b4ddf5ee9e80ee0c1f52401f2456b84ae1ae6a1a5f207", + "sha256:3534e7dcbdcf757da6b85a0bbf5b6868786d5982dd959b065e65481644817a18", + "sha256:36836d6ff945a00b88ba1e4572d721e60b5b8c98c155d465f56ad19d68f23734", + "sha256:38c0109396c4cfc574d502df99742a45c72c08eff0a36158b6f04000043dbf38", + "sha256:3946fa46a0cf3e4c8cb1cc52f56bb536310d34f25f01ca9b6c16afa767dab110", + "sha256:3bec022aec2c514d9cf199522a802bd007cd588ab17ab2525f20f9c34d067c18", + "sha256:3c9a494bc5ec77d43cea229c4f6db1e4d8fe7e1bbffa8b6f0f0032430ff8ab44", + "sha256:3dce51d0f5e7951f8bb4900c257dad282f49190fdbebecd4ba99bcc41fef404d", + "sha256:3dedcc22d73ec993f42055eff4fcfed9318d1eeb9a6606c55892a26964964e48", + "sha256:4042d5c8f957e15221d423ba781e85d553722fc4113f523f2feb7b188cc34c5e", + "sha256:481551899c856c704d58119b5025793fa6730adda3571971af568f66d2424bb5", + "sha256:4dc1e73c36828f982bfe79fadf5919923f8a6f4df2860804db9a98c48824ce8d", + "sha256:4e5163c14bffd570ef2affbfdd77bba66383890797df43dc8b4cc7d6f500bf53", + "sha256:511ef87c8aec0783e08ac18565a16d435372bc1ac25a91e6ac7f5ef2b0bff790", + "sha256:532bc9bf33a68613fd7d65e4b1c71a6a38d7d42604ecf239c77392e9b4e8998c", + "sha256:54523e136b8948060c0fa0bc7b1b50c32c186f2fceee897a495406bb6e311d2b", + "sha256:5649fd1c7bade02f320a462fdefd0b4bd3ce036065836d4f42e0de958038e116", + "sha256:56be790f86bfb2c98fb742ce566dfb4816e5a83384616ab59c49e0604d49c51d", + "sha256:5b77459df20e08151cd6f8b9ef8ef1f961ef73d85c21a555c7eed5b79410ec10", + "sha256:5ed6ab538499c8644b8a3e18debabcd7ce684f3fa91cf867521a7a0279cab2d6", + "sha256:6178f72c5508bfc5fd446a5905e698c6212932f25bcdd4b47a757a50605a90e2", + "sha256:6370e8686f662e6a3941ee48ed4742317cafbe5707e36406e9df792cdb535776", + "sha256:64f02c6841d7d83f832cd97ccf8eb8a906d06eb95d5276069175c696b024b60a", + "sha256:65bcd23054beab4d166035cabbc868a09c1a49d1efe458fe8e4361215df40265", + "sha256:66671f93accb62ed07da56613636f3641f1a12c13046ce91ffc923721f23c008", + "sha256:6696b7688f54f5af4462118f0bfa7c1621eeb87154f77fa04b9295ce7a8f2943", + "sha256:6785f414ae0f3c733c437e0f3929197934f526d19dfaa75e18fdb4f94c6fb374", + "sha256:67f6279d125ca0046a7fd386d01b311c6363844deac3e5b069b514ba3e63c246", + "sha256:6c114670c45346afedc0d947faf3c7f701051d2518b943679c8ff88befe14f8e", + "sha256:6e0d51f618228538a3e8f46bd246f87a6cd030565e015803691603f55e12afb5", + "sha256:6ed74185b2db44f41ef35fd1617c5888e59792da9bbc9190d6c7300617182616", + "sha256:708838739abf24b2ceb208d0e22403dd018faeef86ddac04319a62ae884c4f15", + "sha256:715479b9a2802ecac752a3b0efa2b0b60285cf962ee38414211abdfccc233b41", + "sha256:733784b6d6def852c814bce5f318d25da2ee65dd4839a0718641c696e09a2960", + "sha256:750e02e074872a3fad7f233b47734166440af3cdea0add3e95163110816d6752", + "sha256:752a45dc4a6934060b3b0dab47e04edc3326575f82be64bc4fc293914566503e", + "sha256:7579e913a5339fb8fa133f6bbcfd8e6749696206cf05acdbdca71a1b436d8e72", + "sha256:7641bb8895e77f921102f72833904dcd9901df5d6d72a2ab8f31d04b7e51e4e7", + "sha256:7804338df6fcc08105c7745f1502ba68d900f45fd770d5bdd5288ddccb8a42d8", + "sha256:80d04837f55fc81da168b98de4f4b797ef007fc8a79ab71c6ec9bc4dd662b15b", + "sha256:813c0e0132266c08eb87469a642cb30aaff57c5f426255419572aaeceeaa7bf4", + "sha256:82b271f5137d07749f7bf32f70b17ab6eaabedd297e75dce75081a24f76eb545", + "sha256:84c018e49c3bf790f9c2771c45e9313a08c2c2a6342b162cd650258b57817706", + "sha256:8751d2787c9131302398b11e6c8068053dcb55d5a8964e114b6e196cf16cb366", + "sha256:8778f0c7a52e56f75d12dae53ae320fae900a8b9b4164b981b9c5ce059cd1fcb", + "sha256:87fad7d9ba98c86bcb41b2dc8dbb326619be2562af1f8ff50776a39e55721c5a", + "sha256:8d828b6667a32a728a1ad1d93957cdf37489c57b97ae6c4de2860fa749b8fc1e", + "sha256:8e385e4267ab76874ae30db04c627faaaf0b509e1ccc11a95b3fc3e83f855c00", + "sha256:92a0a01ead5e668468e952e4238cccd7c537364eb7d851ab144ab6627dbbe12f", + "sha256:94e1885b270625a9a828c9793b4d52a64445299baa1fea5a173bf1d3dd9a1a5a", + "sha256:a180c5e59792af262bf263b21a3c49353f25945d8d9f70628e73de370d55e1e1", + "sha256:a277ab8928b9f299723bc1a2dabb1265911b1a76341f90a510368ca44ad9ab66", + "sha256:a5fe03b42827c13cdccd08e6c0247b6a6d4b5e3cdc53fd1749f5896adcdc2356", + "sha256:a6c5863edfbe888d9eff9c8b8087354e27618d9da76425c119293f11712a6319", + "sha256:a89c23ef8d2c6b27fd200a42aa4ac72786e7c60d40efdc76e6011260b6e949c4", + "sha256:adb2597b428735679446b46c8badf467b4ca5f5056aae4d51a19f9570301b1ad", + "sha256:ae196f021b5e7c78e918242d217db021ed2a6ace2bc6ae94c0fc596221c7f58d", + "sha256:ae89db9e5f98a11a4bf50407d4363e7b09b31e55bc117b4f7d80aab97ba009e5", + "sha256:aed52fea0513bac0ccde438c188c8a471c4e0f457c2dd20cdbf6ea7a450046c7", + "sha256:aef65cd602a6d0e0ff6f9930fcb1c8fec60dd2cfcb6facaf4bdb0e5873042db0", + "sha256:af21eb4409a119e365397b2adbaca4c9ccab56543a65d5dbd9f920d6ac29f686", + "sha256:b14b2d9dac08e28bb8046a1a0434b1750eb221c8f5b87a68f4fa11a6f97b5e34", + "sha256:bb6d88045545b26da47aa879dd4a89a71d1dce0f0e549b1abcb31dfe4a8eac49", + "sha256:bb8cc7534f51d9a017b93e3e85b260924f909601c3df002bcdb58ddb4dc41a5c", + "sha256:bc17a677b21b3502a21f66a8cc64f5bfad4df8a0b8434d661666f8ce90ac3af1", + "sha256:bd6c2a1c7573c64738d716488d2cdd3c00e340e4835707d8fdb8dc1a66ef164e", + "sha256:bd9b23791fe793e4968dba0c447e12f78e425c59fc0e3b97f6450f4781f3ee60", + "sha256:c03a41a8784091e67a39648f70c5f97b5b6a37f216896d44d2cdcb82615339a0", + "sha256:c0f081d69a6e58272819b70288d3221a6ee64b98df852631c80f293514d3b274", + "sha256:c35abb8bfff0185efac5878da64c45dafd2b37fb0383add1be155a763c1f083d", + "sha256:c36c333c39be2dbca264d7803333c896ab8fa7d4d6f0ab7edb7dfd7aea6e98c0", + "sha256:c45e9440fb78f8ddabcf714b68f936737a121355bf59f3907f4e17721b9d1aae", + "sha256:c593052c465475e64bbfe5dbd81680f64a67fdc752c56d7a0ae205dc8aeefe0f", + "sha256:cdd68a1fb318e290a2077696b7eb7a21a49163c455979c639bf5a5dcdc46617d", + "sha256:ce3412fbe1e31eb81ea42f4169ed94861c56e643189e1e75f0041f3fe7020abe", + "sha256:cf1493cd8607bec4d8a7b9b004e699fcf8f9103a9284cc94962cb73d20f9d4a3", + "sha256:cf29836da5119f3c8a8a70667b0ef5fdca3bb12f80fd06487cfa575b3909b393", + "sha256:d4a48e5b3c2a489fae013b7589308a40146ee081f6f509e047e0e096084ceca1", + "sha256:d560742f3c0d62afaccf9f41fe485ed69bd7661a241f86a3ef0f0fb8b1a397af", + "sha256:d6038d37043bced98a66e68d3aa2b6a35505dc01328cd65217cefe82f25def44", + "sha256:d61f00a0869d77422d9b2aba989e2d24afa6ffd552af442e0e58de4f35ea6d00", + "sha256:d635aab80466bc95771bb78d5370e74d36d1fe31467b6b29b8b57b2a3cd7d22c", + "sha256:dca4bbc466a95ba9c0234ef56d7dd9509f63da22274589ebd4ed7f1f4d4c54e3", + "sha256:dd915403e231e6b1809fe9b6d9fc55cf8fb5e02765ac625d9cd623342a7905d7", + "sha256:e044c39e41b92c845bc815e5ae4230804e8e7bc29e399b0437d64222d92809dd", + "sha256:e060d01aec0a910bdccb8be71faf34e7799ce36950f8294c8bf612cba65a2c9e", + "sha256:e1421b502d83040e6d7fb2fb18dff63957f720da3d77b2fbd3187ceb63755d7b", + "sha256:e17b8d5d6a8c47c85e68ca8379def1303fd360c3e22093a807cd34a71cd082b8", + "sha256:e5f4d355f0a2b1a31bc3edec6795b46324349c9cb25eed068049e4f472fb4259", + "sha256:e712b419df8ba5e42b226c510472b37bd57b38e897d3eca5e8cfd410a29fa859", + "sha256:e74327fb75de8986940def6e8dee4f127cc9752bee7355bb323cc5b2659b6d46", + "sha256:e80c8378d8f3d83cd3164da1ad2df9e37a666cdde7b1cb2298ed0b558064be30", + "sha256:e8ac484bf18ce6975760921bb6148041faa8fef0547200386ea0b52b5d27bf7b", + "sha256:eca9705049ad3c7345d574e3510665cb2cf844c2f2dcfe675332677f081cbd46", + "sha256:ed065083d0898c9d5b4bbec7b026fd755ff7454e6e8b73a67f8c744b13986e24", + "sha256:edac0f1ab77644605be2cbba52e6b7f630731fc42b34cb0f634be1a6eface56a", + "sha256:effc3f449787117233702311a1b7d8f59cba9ced946ba727bdc329ec69028e24", + "sha256:f22dec1690b584cea26fade98b2435c132c1b5f68e39f5a0b7627cd7ae31f1dc", + "sha256:f495a1652cf3fbab2eb0639776dad966c2fb874d79d87ca07f9d5f059b8bd215", + "sha256:f496c9c3cc02230093d8330875c4c3cdfc3b73612a5fd921c65d39cbcef08063", + "sha256:f59099f9b66f0d7145115e6f80dd8b1d847176df89b234a5a6b3f00437aa0832", + "sha256:f59ad4c0e8f6bba240a9bb85504faa1ab438237199d4cce5f622761507b8f6a6", + "sha256:fbccdc05410c9ee21bbf16a35f4c1d16123dcdeb8a1d38f33654fa21d0234f79", + "sha256:fea24543955a6a729c45a73fe90e08c743f0b3334bbf3201e6c4bc1b0c7fa464" + ], + "markers": "python_version >= '3.7'", + "version": "==3.4.7" + }, + "click": { + "hashes": [ + "sha256:482be17c6991b8c19c5429a1e995d9b0efdbb63172824c41f99965dc0ade8ec2", + "sha256:918b5633eddf6b41c32d4f454bf0de810065c74e3f7dbf8ee5452f8be88d3e96" + ], + "markers": "python_version >= '3.10'", + "version": "==8.4.1" + }, + "colorama": { + "hashes": [ + "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44", + "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6" + ], + "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4, 3.5, 3.6'", + "version": "==0.4.6" + }, + "dnspython": { + "hashes": [ + "sha256:01d9bbc4a2d76bf0db7c1f729812ded6d912bd318d3b1cf81d30c0f845dbf3af", + "sha256:181d3c6996452cb1189c4046c61599b84a5a86e099562ffde77d26984ff26d0f" + ], + "markers": "python_version >= '3.10'", + "version": "==2.8.0" + }, + "editorconfig": { + "hashes": [ + "sha256:1eda9c2c0db8c16dbd50111b710572a5e6de934e39772de1959d41f64fc17c82", + "sha256:23c08b00e8e08cc3adcddb825251c497478df1dada6aefeb01e626ad37303745" + ], + "markers": "python_version >= '3.9'", + "version": "==0.17.1" + }, + "ghp-import": { + "hashes": [ + "sha256:8337dd7b50877f163d4c0289bc1f1c7f127550241988d568c1db512c4324a619", + "sha256:9c535c4c61193c2df8871222567d7fd7e5014d835f97dc7b7439069e2413d343" + ], + "version": "==2.1.0" + }, + "idna": { + "hashes": [ + "sha256:cc246e3a3f89580c3a951b5ad298ca4638078b2cdd4f115654332b5c26daded5", + "sha256:d7a6da03db833450fca25d2358ac9ff06cd624577a4aea3a596d5c0f77b8e03d" + ], + "markers": "python_version >= '3.9'", + "version": "==3.16" + }, + "jinja2": { + "hashes": [ + "sha256:0137fb05990d35f1275a587e9aee6d56da821fc83491a0fb838183be43f66d6d", + "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67" + ], + "markers": "python_version >= '3.7'", + "version": "==3.1.6" + }, + "jsbeautifier": { + "hashes": [ + "sha256:5bb18d9efb9331d825735fbc5360ee8f1aac5e52780042803943aa7f854f7592", + "sha256:72f65de312a3f10900d7685557f84cb61a9733c50dcc27271a39f5b0051bf528" + ], + "version": "==1.15.4" + }, + "linkchecker": { + "hashes": [ + "sha256:5268587ed0b0f7e7521b75905128c96856f30f67dad49f66e2c963bc174ca92d", + "sha256:fb7e8facda7749c2fa5fa5dc241c0adc302da3d31d588964a2570db501aa49e5" + ], + "index": "pypi", + "markers": "python_version >= '3.9'", + "version": "==10.6.0" + }, + "markdown": { + "hashes": [ + "sha256:994d51325d25ad8aa7ce4ebaec003febcce822c3f8c911e3b17c52f7f589f950", + "sha256:e91464b71ae3ee7afd3017d9f358ef0baf158fd9a298db92f1d4761133824c36" + ], + "markers": "python_version >= '3.10'", + "version": "==3.10.2" + }, + "markupsafe": { + "hashes": [ + "sha256:0303439a41979d9e74d18ff5e2dd8c43ed6c6001fd40e5bf2e43f7bd9bbc523f", + "sha256:068f375c472b3e7acbe2d5318dea141359e6900156b5b2ba06a30b169086b91a", + "sha256:0bf2a864d67e76e5c9a34dc26ec616a66b9888e25e7b9460e1c76d3293bd9dbf", + "sha256:0db14f5dafddbb6d9208827849fad01f1a2609380add406671a26386cdf15a19", + "sha256:0eb9ff8191e8498cca014656ae6b8d61f39da5f95b488805da4bb029cccbfbaf", + "sha256:0f4b68347f8c5eab4a13419215bdfd7f8c9b19f2b25520968adfad23eb0ce60c", + "sha256:1085e7fbddd3be5f89cc898938f42c0b3c711fdcb37d75221de2666af647c175", + "sha256:116bb52f642a37c115f517494ea5feb03889e04df47eeff5b130b1808ce7c219", + "sha256:12c63dfb4a98206f045aa9563db46507995f7ef6d83b2f68eda65c307c6829eb", + "sha256:133a43e73a802c5562be9bbcd03d090aa5a1fe899db609c29e8c8d815c5f6de6", + "sha256:1353ef0c1b138e1907ae78e2f6c63ff67501122006b0f9abad68fda5f4ffc6ab", + "sha256:15d939a21d546304880945ca1ecb8a039db6b4dc49b2c5a400387cdae6a62e26", + "sha256:177b5253b2834fe3678cb4a5f0059808258584c559193998be2601324fdeafb1", + "sha256:1872df69a4de6aead3491198eaf13810b565bdbeec3ae2dc8780f14458ec73ce", + "sha256:1b4b79e8ebf6b55351f0d91fe80f893b4743f104bff22e90697db1590e47a218", + "sha256:1b52b4fb9df4eb9ae465f8d0c228a00624de2334f216f178a995ccdcf82c4634", + "sha256:1ba88449deb3de88bd40044603fafffb7bc2b055d626a330323a9ed736661695", + "sha256:1cc7ea17a6824959616c525620e387f6dd30fec8cb44f649e31712db02123dad", + "sha256:218551f6df4868a8d527e3062d0fb968682fe92054e89978594c28e642c43a73", + "sha256:26a5784ded40c9e318cfc2bdb30fe164bdb8665ded9cd64d500a34fb42067b1c", + "sha256:2713baf880df847f2bece4230d4d094280f4e67b1e813eec43b4c0e144a34ffe", + "sha256:2a15a08b17dd94c53a1da0438822d70ebcd13f8c3a95abe3a9ef9f11a94830aa", + "sha256:2f981d352f04553a7171b8e44369f2af4055f888dfb147d55e42d29e29e74559", + "sha256:32001d6a8fc98c8cb5c947787c5d08b0a50663d139f1305bac5885d98d9b40fa", + "sha256:3524b778fe5cfb3452a09d31e7b5adefeea8c5be1d43c4f810ba09f2ceb29d37", + "sha256:3537e01efc9d4dccdf77221fb1cb3b8e1a38d5428920e0657ce299b20324d758", + "sha256:35add3b638a5d900e807944a078b51922212fb3dedb01633a8defc4b01a3c85f", + "sha256:38664109c14ffc9e7437e86b4dceb442b0096dfe3541d7864d9cbe1da4cf36c8", + "sha256:3a7e8ae81ae39e62a41ec302f972ba6ae23a5c5396c8e60113e9066ef893da0d", + "sha256:3b562dd9e9ea93f13d53989d23a7e775fdfd1066c33494ff43f5418bc8c58a5c", + "sha256:457a69a9577064c05a97c41f4e65148652db078a3a509039e64d3467b9e7ef97", + "sha256:4bd4cd07944443f5a265608cc6aab442e4f74dff8088b0dfc8238647b8f6ae9a", + "sha256:4e885a3d1efa2eadc93c894a21770e4bc67899e3543680313b09f139e149ab19", + "sha256:4faffd047e07c38848ce017e8725090413cd80cbc23d86e55c587bf979e579c9", + "sha256:509fa21c6deb7a7a273d629cf5ec029bc209d1a51178615ddf718f5918992ab9", + "sha256:5678211cb9333a6468fb8d8be0305520aa073f50d17f089b5b4b477ea6e67fdc", + "sha256:591ae9f2a647529ca990bc681daebdd52c8791ff06c2bfa05b65163e28102ef2", + "sha256:5a7d5dc5140555cf21a6fefbdbf8723f06fcd2f63ef108f2854de715e4422cb4", + "sha256:69c0b73548bc525c8cb9a251cddf1931d1db4d2258e9599c28c07ef3580ef354", + "sha256:6b5420a1d9450023228968e7e6a9ce57f65d148ab56d2313fcd589eee96a7a50", + "sha256:722695808f4b6457b320fdc131280796bdceb04ab50fe1795cd540799ebe1698", + "sha256:729586769a26dbceff69f7a7dbbf59ab6572b99d94576a5592625d5b411576b9", + "sha256:77f0643abe7495da77fb436f50f8dab76dbc6e5fd25d39589a0f1fe6548bfa2b", + "sha256:795e7751525cae078558e679d646ae45574b47ed6e7771863fcc079a6171a0fc", + "sha256:7be7b61bb172e1ed687f1754f8e7484f1c8019780f6f6b0786e76bb01c2ae115", + "sha256:7c3fb7d25180895632e5d3148dbdc29ea38ccb7fd210aa27acbd1201a1902c6e", + "sha256:7e68f88e5b8799aa49c85cd116c932a1ac15caaa3f5db09087854d218359e485", + "sha256:83891d0e9fb81a825d9a6d61e3f07550ca70a076484292a70fde82c4b807286f", + "sha256:8485f406a96febb5140bfeca44a73e3ce5116b2501ac54fe953e488fb1d03b12", + "sha256:8709b08f4a89aa7586de0aadc8da56180242ee0ada3999749b183aa23df95025", + "sha256:8f71bc33915be5186016f675cd83a1e08523649b0e33efdb898db577ef5bb009", + "sha256:915c04ba3851909ce68ccc2b8e2cd691618c4dc4c4232fb7982bca3f41fd8c3d", + "sha256:949b8d66bc381ee8b007cd945914c721d9aba8e27f71959d750a46f7c282b20b", + "sha256:94c6f0bb423f739146aec64595853541634bde58b2135f27f61c1ffd1cd4d16a", + "sha256:9a1abfdc021a164803f4d485104931fb8f8c1efd55bc6b748d2f5774e78b62c5", + "sha256:9b79b7a16f7fedff2495d684f2b59b0457c3b493778c9eed31111be64d58279f", + "sha256:a320721ab5a1aba0a233739394eb907f8c8da5c98c9181d1161e77a0c8e36f2d", + "sha256:a4afe79fb3de0b7097d81da19090f4df4f8d3a2b3adaa8764138aac2e44f3af1", + "sha256:ad2cf8aa28b8c020ab2fc8287b0f823d0a7d8630784c31e9ee5edea20f406287", + "sha256:b8512a91625c9b3da6f127803b166b629725e68af71f8184ae7e7d54686a56d6", + "sha256:bc51efed119bc9cfdf792cdeaa4d67e8f6fcccab66ed4bfdd6bde3e59bfcbb2f", + "sha256:bdc919ead48f234740ad807933cdf545180bfbe9342c2bb451556db2ed958581", + "sha256:bdd37121970bfd8be76c5fb069c7751683bdf373db1ed6c010162b2a130248ed", + "sha256:be8813b57049a7dc738189df53d69395eba14fb99345e0a5994914a3864c8a4b", + "sha256:c0c0b3ade1c0b13b936d7970b1d37a57acde9199dc2aecc4c336773e1d86049c", + "sha256:c47a551199eb8eb2121d4f0f15ae0f923d31350ab9280078d1e5f12b249e0026", + "sha256:c4ffb7ebf07cfe8931028e3e4c85f0357459a3f9f9490886198848f4fa002ec8", + "sha256:ccfcd093f13f0f0b7fdd0f198b90053bf7b2f02a3927a30e63f3ccc9df56b676", + "sha256:d2ee202e79d8ed691ceebae8e0486bd9a2cd4794cec4824e1c99b6f5009502f6", + "sha256:d53197da72cc091b024dd97249dfc7794d6a56530370992a5e1a08983ad9230e", + "sha256:d6dd0be5b5b189d31db7cda48b91d7e0a9795f31430b7f271219ab30f1d3ac9d", + "sha256:d88b440e37a16e651bda4c7c2b930eb586fd15ca7406cb39e211fcff3bf3017d", + "sha256:de8a88e63464af587c950061a5e6a67d3632e36df62b986892331d4620a35c01", + "sha256:df2449253ef108a379b8b5d6b43f4b1a8e81a061d6537becd5582fba5f9196d7", + "sha256:e1c1493fb6e50ab01d20a22826e57520f1284df32f2d8601fdd90b6304601419", + "sha256:e1cf1972137e83c5d4c136c43ced9ac51d0e124706ee1c8aa8532c1287fa8795", + "sha256:e2103a929dfa2fcaf9bb4e7c091983a49c9ac3b19c9061b6d5427dd7d14d81a1", + "sha256:e56b7d45a839a697b5eb268c82a71bd8c7f6c94d6fd50c3d577fa39a9f1409f5", + "sha256:e8afc3f2ccfa24215f8cb28dcf43f0113ac3c37c2f0f0806d8c70e4228c5cf4d", + "sha256:e8fc20152abba6b83724d7ff268c249fa196d8259ff481f3b1476383f8f24e42", + "sha256:eaa9599de571d72e2daf60164784109f19978b327a3910d3e9de8c97b5b70cfe", + "sha256:ec15a59cf5af7be74194f7ab02d0f59a62bdcf1a537677ce67a2537c9b87fcda", + "sha256:f190daf01f13c72eac4efd5c430a8de82489d9cff23c364c3ea822545032993e", + "sha256:f34c41761022dd093b4b6896d4810782ffbabe30f2d443ff5f083e0cbbb8c737", + "sha256:f3e98bb3798ead92273dc0e5fd0f31ade220f59a266ffd8a4f6065e0a3ce0523", + "sha256:f42d0984e947b8adf7dd6dde396e720934d12c506ce84eea8476409563607591", + "sha256:f71a396b3bf33ecaa1626c255855702aca4d3d9fea5e051b41ac59a9c1c41edc", + "sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a", + "sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50" + ], + "markers": "python_version >= '3.9'", + "version": "==3.0.3" + }, + "mergedeep": { + "hashes": [ + "sha256:0096d52e9dad9939c3d975a774666af186eda617e6ca84df4c94dec30004f2a8", + "sha256:70775750742b25c0d8f36c55aed03d24c3384d17c951b3175d898bd778ef0307" + ], + "markers": "python_version >= '3.6'", + "version": "==1.3.4" + }, + "mkdocs": { + "hashes": [ + "sha256:7b432f01d928c084353ab39c57282f29f92136665bdd6abf7c1ec8d822ef86f2", + "sha256:db91759624d1647f3f34aa0c3f327dd2601beae39a366d6e064c03468d35c20e" + ], + "index": "pypi", + "markers": "python_version >= '3.8'", + "version": "==1.6.1" + }, + "mkdocs-get-deps": { + "hashes": [ + "sha256:8ee8d5f316cdbbb2834bc1df6e69c08fe769a83e040060de26d3c19fad3599a1", + "sha256:e7878cbeac04860b8b5e0ca31d3abad3df9411a75a32cde82f8e44b6c16ff650" + ], + "markers": "python_version >= '3.9'", + "version": "==0.2.2" + }, + "mkdocs-material": { + "hashes": [ + "sha256:00bdde50574f776d328b1862fe65daeaf581ec309bd150f7bff345a098c64a69", + "sha256:71b84353921b8ea1ba84fe11c50912cc512da8fe0881038fcc9a0761c0e635ba" + ], + "index": "pypi", + "markers": "python_version >= '3.8'", + "version": "==9.7.6" + }, + "mkdocs-material-extensions": { + "hashes": [ + "sha256:10c9511cea88f568257f960358a467d12b970e1f7b2c0e5fb2bb48cab1928443", + "sha256:adff8b62700b25cb77b53358dad940f3ef973dd6db797907c49e3c2ef3ab4e31" + ], + "markers": "python_version >= '3.8'", + "version": "==1.3.1" + }, + "mkdocs-mermaid2-plugin": { + "hashes": [ + "sha256:33f60c582be623ed53829a96e19284fc7f1b74a1dbae78d4d2e47fe00c3e190d", + "sha256:fb6f901d53e5191e93db78f93f219cad926ccc4d51e176271ca5161b6cc5368c" + ], + "index": "pypi", + "markers": "python_version >= '3.8'", + "version": "==1.2.3" + }, + "packaging": { + "hashes": [ + "sha256:29572ef2b1f17581046b3a2227d5c611fb25ec70ca1ba8554b24b0e69331a484", + "sha256:d443872c98d677bf60f6a1f2f8c1cb748e8fe762d2bf9d3148b5599295b0fc4f" + ], + "markers": "python_version >= '3.8'", + "version": "==25.0" + }, + "paginate": { + "hashes": [ + "sha256:22bd083ab41e1a8b4f3690544afb2c60c25e5c9a63a30fa2f483f6c60c8e5945", + "sha256:b885e2af73abcf01d9559fd5216b57ef722f8c42affbb63942377668e35c7591" + ], + "version": "==0.5.7" + }, + "pathspec": { + "hashes": [ + "sha256:17db5ecd524104a120e173814c90367a96a98d07c45b2e10c2f3919fff91bf5a", + "sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189" + ], + "markers": "python_version >= '3.9'", + "version": "==1.1.1" + }, + "platformdirs": { + "hashes": [ + "sha256:3bfa75b0ad0db84096ae777218481852c0ebc6c727b3168c1b9e0118e458cf0a", + "sha256:e61adb1d5e5cb3441b4b7710bea7e4c12250ca49439228cc1021c00dcfac0917" + ], + "markers": "python_version >= '3.10'", + "version": "==4.9.6" + }, + "pygments": { + "hashes": [ + "sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f", + "sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176" + ], + "markers": "python_version >= '3.9'", + "version": "==2.20.0" + }, + "pymdown-extensions": { + "hashes": [ + "sha256:72cfcf55f07aea0d4af2c4f11dd4e52466ddfb1bb819673146398e0bd3a77354", + "sha256:d7a5d08014fc571e80ca21dd6f854e31f94c489800350564d55d15b3c41e76b6" + ], + "markers": "python_version >= '3.9'", + "version": "==10.21.3" + }, + "python-dateutil": { + "hashes": [ + "sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3", + "sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427" + ], + "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2'", + "version": "==2.9.0.post0" + }, + "pyyaml": { + "hashes": [ + "sha256:00c4bdeba853cc34e7dd471f16b4114f4162dc03e6b7afcc2128711f0eca823c", + "sha256:0150219816b6a1fa26fb4699fb7daa9caf09eb1999f3b70fb6e786805e80375a", + "sha256:02893d100e99e03eda1c8fd5c441d8c60103fd175728e23e431db1b589cf5ab3", + "sha256:02ea2dfa234451bbb8772601d7b8e426c2bfa197136796224e50e35a78777956", + "sha256:0f29edc409a6392443abf94b9cf89ce99889a1dd5376d94316ae5145dfedd5d6", + "sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c", + "sha256:16249ee61e95f858e83976573de0f5b2893b3677ba71c9dd36b9cf8be9ac6d65", + "sha256:1d37d57ad971609cf3c53ba6a7e365e40660e3be0e5175fa9f2365a379d6095a", + "sha256:1ebe39cb5fc479422b83de611d14e2c0d3bb2a18bbcb01f229ab3cfbd8fee7a0", + "sha256:214ed4befebe12df36bcc8bc2b64b396ca31be9304b8f59e25c11cf94a4c033b", + "sha256:2283a07e2c21a2aa78d9c4442724ec1eb15f5e42a723b99cb3d822d48f5f7ad1", + "sha256:22ba7cfcad58ef3ecddc7ed1db3409af68d023b7f940da23c6c2a1890976eda6", + "sha256:27c0abcb4a5dac13684a37f76e701e054692a9b2d3064b70f5e4eb54810553d7", + "sha256:28c8d926f98f432f88adc23edf2e6d4921ac26fb084b028c733d01868d19007e", + "sha256:2e71d11abed7344e42a8849600193d15b6def118602c4c176f748e4583246007", + "sha256:34d5fcd24b8445fadc33f9cf348c1047101756fd760b4dacb5c3e99755703310", + "sha256:37503bfbfc9d2c40b344d06b2199cf0e96e97957ab1c1b546fd4f87e53e5d3e4", + "sha256:3c5677e12444c15717b902a5798264fa7909e41153cdf9ef7ad571b704a63dd9", + "sha256:3ff07ec89bae51176c0549bc4c63aa6202991da2d9a6129d7aef7f1407d3f295", + "sha256:41715c910c881bc081f1e8872880d3c650acf13dfa8214bad49ed4cede7c34ea", + "sha256:418cf3f2111bc80e0933b2cd8cd04f286338bb88bdc7bc8e6dd775ebde60b5e0", + "sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e", + "sha256:4a2e8cebe2ff6ab7d1050ecd59c25d4c8bd7e6f400f5f82b96557ac0abafd0ac", + "sha256:4ad1906908f2f5ae4e5a8ddfce73c320c2a1429ec52eafd27138b7f1cbe341c9", + "sha256:501a031947e3a9025ed4405a168e6ef5ae3126c59f90ce0cd6f2bfc477be31b7", + "sha256:5190d403f121660ce8d1d2c1bb2ef1bd05b5f68533fc5c2ea899bd15f4399b35", + "sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb", + "sha256:5cf4e27da7e3fbed4d6c3d8e797387aaad68102272f8f9752883bc32d61cb87b", + "sha256:5e0b74767e5f8c593e8c9b5912019159ed0533c70051e9cce3e8b6aa699fcd69", + "sha256:5ed875a24292240029e4483f9d4a4b8a1ae08843b9c54f43fcc11e404532a8a5", + "sha256:5fcd34e47f6e0b794d17de1b4ff496c00986e1c83f7ab2fb8fcfe9616ff7477b", + "sha256:5fdec68f91a0c6739b380c83b951e2c72ac0197ace422360e6d5a959d8d97b2c", + "sha256:6344df0d5755a2c9a276d4473ae6b90647e216ab4757f8426893b5dd2ac3f369", + "sha256:64386e5e707d03a7e172c0701abfb7e10f0fb753ee1d773128192742712a98fd", + "sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824", + "sha256:66291b10affd76d76f54fad28e22e51719ef9ba22b29e1d7d03d6777a9174198", + "sha256:66e1674c3ef6f541c35191caae2d429b967b99e02040f5ba928632d9a7f0f065", + "sha256:6adc77889b628398debc7b65c073bcb99c4a0237b248cacaf3fe8a557563ef6c", + "sha256:79005a0d97d5ddabfeeea4cf676af11e647e41d81c9a7722a193022accdb6b7c", + "sha256:7c6610def4f163542a622a73fb39f534f8c101d690126992300bf3207eab9764", + "sha256:7f047e29dcae44602496db43be01ad42fc6f1cc0d8cd6c83d342306c32270196", + "sha256:8098f252adfa6c80ab48096053f512f2321f0b998f98150cea9bd23d83e1467b", + "sha256:850774a7879607d3a6f50d36d04f00ee69e7fc816450e5f7e58d7f17f1ae5c00", + "sha256:8d1fab6bb153a416f9aeb4b8763bc0f22a5586065f86f7664fc23339fc1c1fac", + "sha256:8da9669d359f02c0b91ccc01cac4a67f16afec0dac22c2ad09f46bee0697eba8", + "sha256:8dc52c23056b9ddd46818a57b78404882310fb473d63f17b07d5c40421e47f8e", + "sha256:9149cad251584d5fb4981be1ecde53a1ca46c891a79788c0df828d2f166bda28", + "sha256:93dda82c9c22deb0a405ea4dc5f2d0cda384168e466364dec6255b293923b2f3", + "sha256:96b533f0e99f6579b3d4d4995707cf36df9100d67e0c8303a0c55b27b5f99bc5", + "sha256:9c57bb8c96f6d1808c030b1687b9b5fb476abaa47f0db9c0101f5e9f394e97f4", + "sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b", + "sha256:9f3bfb4965eb874431221a3ff3fdcddc7e74e3b07799e0e84ca4a0f867d449bf", + "sha256:a33284e20b78bd4a18c8c2282d549d10bc8408a2a7ff57653c0cf0b9be0afce5", + "sha256:a80cb027f6b349846a3bf6d73b5e95e782175e52f22108cfa17876aaeff93702", + "sha256:b30236e45cf30d2b8e7b3e85881719e98507abed1011bf463a8fa23e9c3e98a8", + "sha256:b3bc83488de33889877a0f2543ade9f70c67d66d9ebb4ac959502e12de895788", + "sha256:b865addae83924361678b652338317d1bd7e79b1f4596f96b96c77a5a34b34da", + "sha256:b8bb0864c5a28024fac8a632c443c87c5aa6f215c0b126c449ae1a150412f31d", + "sha256:ba1cc08a7ccde2d2ec775841541641e4548226580ab850948cbfda66a1befcdc", + "sha256:bdb2c67c6c1390b63c6ff89f210c8fd09d9a1217a465701eac7316313c915e4c", + "sha256:c1ff362665ae507275af2853520967820d9124984e0f7466736aea23d8611fba", + "sha256:c2514fceb77bc5e7a2f7adfaa1feb2fb311607c9cb518dbc378688ec73d8292f", + "sha256:c3355370a2c156cffb25e876646f149d5d68f5e0a3ce86a5084dd0b64a994917", + "sha256:c458b6d084f9b935061bc36216e8a69a7e293a2f1e68bf956dcd9e6cbcd143f5", + "sha256:d0eae10f8159e8fdad514efdc92d74fd8d682c933a6dd088030f3834bc8e6b26", + "sha256:d76623373421df22fb4cf8817020cbb7ef15c725b9d5e45f17e189bfc384190f", + "sha256:ebc55a14a21cb14062aa4162f906cd962b28e2e9ea38f9b4391244cd8de4ae0b", + "sha256:eda16858a3cab07b80edaf74336ece1f986ba330fdb8ee0d6c0d68fe82bc96be", + "sha256:ee2922902c45ae8ccada2c5b501ab86c36525b883eff4255313a253a3160861c", + "sha256:efd7b85f94a6f21e4932043973a7ba2613b059c4a000551892ac9f1d11f5baf3", + "sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6", + "sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926", + "sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0" + ], + "index": "pypi", + "markers": "python_version >= '3.8'", + "version": "==6.0.3" + }, + "pyyaml-env-tag": { + "hashes": [ + "sha256:17109e1a528561e32f026364712fee1264bc2ea6715120891174ed1b980d2e04", + "sha256:2eb38b75a2d21ee0475d6d97ec19c63287a7e140231e4214969d0eac923cd7ff" + ], + "markers": "python_version >= '3.9'", + "version": "==1.1" + }, + "requests": { + "hashes": [ + "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0", + "sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed" + ], + "markers": "python_version >= '3.10'", + "version": "==2.34.2" + }, + "setuptools": { + "hashes": [ + "sha256:487b53915f52501f0a79ccfd0c02c165ffe06631443a886740b91af4b7a5845a", + "sha256:fdd925d5c5d9f62e4b74b30d6dd7828ce236fd6ed998a08d81de62ce5a6310d6" + ], + "markers": "python_version >= '3.9'", + "version": "==81.0.0" + }, + "six": { + "hashes": [ + "sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274", + "sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81" + ], + "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2'", + "version": "==1.17.0" + }, + "soupsieve": { + "hashes": [ + "sha256:3267f1eeea4251fb42728b6dfb746edc9acaffc4a45b27e19450b676586e8349", + "sha256:ed64f2ba4eebeab06cc4962affce381647455978ffc1e36bb79a545b91f45a95" + ], + "markers": "python_version >= '3.9'", + "version": "==2.8.3" + }, + "typing-extensions": { + "hashes": [ + "sha256:0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466", + "sha256:f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548" + ], + "markers": "python_version >= '3.9'", + "version": "==4.15.0" + }, + "urllib3": { + "hashes": [ + "sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c", + "sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897" + ], + "markers": "python_version >= '3.10'", + "version": "==2.7.0" + }, + "watchdog": { + "hashes": [ + "sha256:07df1fdd701c5d4c8e55ef6cf55b8f0120fe1aef7ef39a1c6fc6bc2e606d517a", + "sha256:20ffe5b202af80ab4266dcd3e91aae72bf2da48c0d33bdb15c66658e685e94e2", + "sha256:212ac9b8bf1161dc91bd09c048048a95ca3a4c4f5e5d4a7d1b1a7d5752a7f96f", + "sha256:2cce7cfc2008eb51feb6aab51251fd79b85d9894e98ba847408f662b3395ca3c", + "sha256:490ab2ef84f11129844c23fb14ecf30ef3d8a6abafd3754a6f75ca1e6654136c", + "sha256:6eb11feb5a0d452ee41f824e271ca311a09e250441c262ca2fd7ebcf2461a06c", + "sha256:6f10cb2d5902447c7d0da897e2c6768bca89174d0c6e1e30abec5421af97a5b0", + "sha256:7607498efa04a3542ae3e05e64da8202e58159aa1fa4acddf7678d34a35d4f13", + "sha256:76aae96b00ae814b181bb25b1b98076d5fc84e8a53cd8885a318b42b6d3a5134", + "sha256:7a0e56874cfbc4b9b05c60c8a1926fedf56324bb08cfbc188969777940aef3aa", + "sha256:82dc3e3143c7e38ec49d61af98d6558288c415eac98486a5c581726e0737c00e", + "sha256:9041567ee8953024c83343288ccc458fd0a2d811d6a0fd68c4c22609e3490379", + "sha256:90c8e78f3b94014f7aaae121e6b909674df5b46ec24d6bebc45c44c56729af2a", + "sha256:9513f27a1a582d9808cf21a07dae516f0fab1cf2d7683a742c498b93eedabb11", + "sha256:9ddf7c82fda3ae8e24decda1338ede66e1c99883db93711d8fb941eaa2d8c282", + "sha256:a175f755fc2279e0b7312c0035d52e27211a5bc39719dd529625b1930917345b", + "sha256:a1914259fa9e1454315171103c6a30961236f508b9b623eae470268bbcc6a22f", + "sha256:afd0fe1b2270917c5e23c2a65ce50c2a4abb63daafb0d419fde368e272a76b7c", + "sha256:bc64ab3bdb6a04d69d4023b29422170b74681784ffb9463ed4870cf2f3e66112", + "sha256:bdd4e6f14b8b18c334febb9c4425a878a2ac20efd1e0b231978e7b150f92a948", + "sha256:c7ac31a19f4545dd92fc25d200694098f42c9a8e391bc00bdd362c5736dbf881", + "sha256:c7c15dda13c4eb00d6fb6fc508b3c0ed88b9d5d374056b239c4ad1611125c860", + "sha256:c897ac1b55c5a1461e16dae288d22bb2e412ba9807df8397a635d88f671d36c3", + "sha256:cbafb470cf848d93b5d013e2ecb245d4aa1c8fd0504e863ccefa32445359d680", + "sha256:d1cdb490583ebd691c012b3d6dae011000fe42edb7a82ece80965b42abd61f26", + "sha256:e3df4cbb9a450c6d49318f6d14f4bbc80d763fa587ba46ec86f99f9e6876bb26", + "sha256:e6439e374fc012255b4ec786ae3c4bc838cd7309a540e5fe0952d03687d8804e", + "sha256:e6f0e77c9417e7cd62af82529b10563db3423625c5fce018430b249bf977f9e8", + "sha256:e7631a77ffb1f7d2eefa4445ebbee491c720a5661ddf6df3498ebecae5ed375c", + "sha256:ef810fbf7b781a5a593894e4f439773830bdecb885e6880d957d5b9382a960d2" + ], + "markers": "python_version >= '3.9'", + "version": "==6.0.0" + } + } } diff --git a/README.md b/README.md index 17e48c5..95b8991 100644 --- a/README.md +++ b/README.md @@ -4,6 +4,8 @@ A local-first research pipeline that retrieves academic papers from multiple sch Built with Python 3.13, pydantic-ai, sentence-transformers, and async I/O. +**Documentation:** [https://ndevu12.github.io/Research_Assistant_Model/](https://ndevu12.github.io/Research_Assistant_Model/) — architecture, configuration, API, operations, and known issues. + ## Features - **Multi-stage pipeline** — query understanding → expansion → retrieval → deduplication → ranking → clustering → synthesis → gap analysis → citation export → report generation diff --git a/contributing.md b/contributing.md new file mode 100644 index 0000000..d386dfe --- /dev/null +++ b/contributing.md @@ -0,0 +1,75 @@ +# Contributing + +Thank you for improving the AI Research Assistant. This file is the **canonical contributor entry** for GitHub — the docs site links here rather than duplicating it. + +**Full documentation:** https://ndevu12.github.io/Research_Assistant_Model/ + +## Code contributions + +### Setup + +```bash +pipenv install --dev +pipenv run pytest +``` + +See the docs site for [local development setup](https://ndevu12.github.io/Research_Assistant_Model/development/local-setup/) and [testing](https://ndevu12.github.io/Research_Assistant_Model/development/testing/). + +### Guidelines + +1. Run `pipenv run pytest -m "not slow"` before opening a PR. +2. Match existing import and module layout — see [import conventions](https://ndevu12.github.io/Research_Assistant_Model/development/import-conventions/). +3. Keep changes focused; avoid unrelated refactors in the same PR. +4. Add or update tests when behavior changes. + +## Documentation contributions + +Documentation is published with [MkDocs Material](https://squidfunk.github.io/mkdocs-material/) from the `docs/` directory. + +### Local preview + +```bash +pipenv install --dev +pipenv run mkdocs serve +# → http://127.0.0.1:8000/Research_Assistant_Model/ +``` + +Build without serving (same checks as CI): + +```bash +pipenv run mkdocs build --strict +pipenv run python scripts/check_docs_policy.py +``` + +### File location policy + +| Location | Role | +|----------|------| +| Root `README.md`, `contributing.md` | GitHub landing and contributor entry — **stay at repo root** | +| `docs/` | Deep, code-backed reference pages for the docs site | +| `docs/contributing.md` | Short pointer to this file only | + +**Do:** write new deep content in `docs/` from code analysis; link getting-started pages to the root README for copy-paste commands. + +**Do not:** move or gut root README/contributing; copy-paste README body into docs pages; ship scaffold placeholders (CI rejects them). + +### Writing conventions + +1. **Reference pattern:** Docs pages link to the root README for quick-start commands; add internals and analysis below the link. +2. **Admonition types:** `warning` for stubs/known bugs, `tip` for cookbooks, `info` for defaults. +3. **Code paths:** Use `pipenv run python -m src` in examples. +4. **Config examples:** Show YAML + equivalent `RA_*` env override side-by-side. +5. **Mermaid:** Use for pipeline/LLM diagrams; keep node IDs camelCase. +6. **Status tags:** Mark API and stub providers as experimental or planned. + +Authoritative references: + +| Topic | Docs page | +|-------|-----------| +| Canonical sources (command blocks, warnings) | [reference/canonical-sources.md](https://ndevu12.github.io/Research_Assistant_Model/reference/canonical-sources/) | +| Known issues | [quality/known-issues.md](https://ndevu12.github.io/Research_Assistant_Model/quality/known-issues/) | +| Setup system | [setup-system/index.md](https://ndevu12.github.io/Research_Assistant_Model/setup-system/) | +| Publishing / CI | [development/publishing.md](https://ndevu12.github.io/Research_Assistant_Model/development/publishing/) | +| Environment variables | [configuration/environment-variables.md](https://ndevu12.github.io/Research_Assistant_Model/configuration/environment-variables/) | + +The legacy file `docs/research-quality-known-issues.md` is a redirect only; edit `docs/quality/known-issues.md`. diff --git a/docs/README.md b/docs/README.md deleted file mode 100644 index 7c84450..0000000 --- a/docs/README.md +++ /dev/null @@ -1,12 +0,0 @@ -# Documentation - -Technical notes and design records for the AI Research Assistant. - -| Document | Description | -|----------|-------------| -| [Research Quality — Known Issues](./research-quality-known-issues.md) | Analysis of poor/off-topic reports (May 2026), root causes, and planned fixes | - -## Conventions - -- **Known issues** docs describe verified problems with reproduction context, code references, and a prioritized fix backlog. -- Implementation status in those docs is updated when fixes land; until then, treat recommended fixes as **planned**, not shipped. diff --git a/docs/_analysis/README.md b/docs/_analysis/README.md new file mode 100644 index 0000000..6e82080 --- /dev/null +++ b/docs/_analysis/README.md @@ -0,0 +1,23 @@ +# Phase 0 — Code Analysis Artifacts + +Internal reference produced from source and test analysis. **Not published** in the MkDocs site nav — use these artifacts when writing user-facing pages in Phase 1+. + +| Artifact | Purpose | Primary consumers | +|----------|---------|-------------------| +| [artifact-registry.md](artifact-registry.md) | Pipeline stage I/O, artifacts, config keys, LLM usage | `architecture/pipeline-stages.md`, `architecture/overview.md`, `configuration/stage-toggles.md` | +| [config-inventory.md](config-inventory.md) | `AppSettings` fields, YAML files, env vars, precedence | `configuration/*`, `getting-started/*` | +| [test-behavior-index.md](test-behavior-index.md) | Test map, mocks, coverage gaps | `development/testing.md` | +| [provider-http-matrix.md](provider-http-matrix.md) | Retrieval provider HTTP details, CLI vs pipeline | `retrieval/*`, `operations/troubleshooting.md` | +| [llm-resolution-tree.md](llm-resolution-tree.md) | LLM provider/model/feature resolution | `llm/*`, `architecture/llm-layer.md` | + +**Generated:** 2026-05-22 +**Source revision:** analyzed against current `main` tree (`src/`, `config/`, `tests/`). + +## Key findings (executive summary) + +1. **11 pipeline stages** share artifacts via `PipelineContext`; final `ResearchPipelineResult.artifacts` exports a subset (see artifact registry). +2. **CLI shortcut** (`run_research_helper`) hardcodes OpenAlex + Semantic Scholar — differs from full pipeline/API. +3. **Heuristic LLM defaults** (`synthesis.llm_enabled: false`, `query_expansion.llm_enabled: false`) resolved at pipeline start via `resolve_effective_settings()`. +4. **3 retrieval stubs** (PubMed, CORE, DBLP) raise `NotImplementedError` if enabled. +5. **Config precedence:** constructor kwargs > process env > `.env` > merged YAML > field defaults; plus post-load LLM feature resolution. +6. **Doc/code gaps to fix in Phase 1:** `llm.timeout_seconds` and `llm.temperature` unused; per-provider `limit` ignored by `RetrievalStage`; `.env.example` enables `RA_DEBUG=1` while `RA_PIPELINE__DEBUG=false`. diff --git a/docs/_analysis/artifact-registry.md b/docs/_analysis/artifact-registry.md new file mode 100644 index 0000000..f7eb06b --- /dev/null +++ b/docs/_analysis/artifact-registry.md @@ -0,0 +1,255 @@ +# Artifact Registry — Pipeline Stages + +Source: `src/core/pipeline.py`, `src/core/registry.py`, `src/retrieval/orchestrator.py`, stage modules under `src/research/`, `src/retrieval/`, `src/analysis/`, `src/reporting/`. + +## Stage order + +``` +query_understanding → query_expansion → retrieval → deduplication → ranking → +relevance_scoring → clustering → synthesis → gap_analysis → citation_export → report_generation +``` + +Built by `build_pipeline()` in `src/retrieval/orchestrator.py:37–50`. Registry keys in `src/core/registry.py:136–148`. + +## Execution model + +| Mechanism | Location | Behavior | +|-----------|----------|----------| +| Sequential `data` chain | `pipeline.py:148` | Each stage output becomes next stage input | +| Shared artifact store | `context.py` | `ctx.get_artifact` / `ctx.set_artifact` | +| Stage enable gate | `pipeline.py` + `settings.pipeline.enabled_stages` | Disabled stages skipped | +| Timeouts | `pipeline.py:66–69` | Default `stage_timeout_seconds` (300s); synthesis uses `synthesis_timeout_seconds` (600s) | +| LLM resolution | `pipeline.py:130` | `resolve_effective_settings()` before stages run | +| Failure handling | `continue_on_stage_failure` | Default `True`; partial results + warnings | +| Debug dump | `pipeline.py:154–155` | When `debug_enabled`, writes `logs/debug/pipeline_*.json` | + +## Master artifact map + +| Artifact key | Set by | Read by | +|--------------|--------|---------| +| `query_understanding` | query_understanding | relevance_scoring | +| `expanded_queries` | query_expansion | — | +| `cached_papers` | orchestrator (`initial_artifacts`) | retrieval | +| `retrieved_papers` | retrieval | synthesis (recovery) | +| `deduplication_stats` | deduplication | — | +| `query_embedding` | ranking (`embedding_context.py`) | relevance_scoring, clustering | +| `paper_embeddings` | ranking (`embedding_context.py`) | relevance_scoring, clustering | +| `ranked_papers` | ranking, relevance_scoring, synthesis | synthesis, citation_export, report_generation | +| `relevance_filter_reasons` | relevance_scoring | — | +| `paper_clusters` | clustering | synthesis, gap_analysis, report_generation | +| `paper_extractions` | synthesis | synthesis (recovery) | +| `paper_analyses` | synthesis | report_generation | +| `synthesis_result` | synthesis | gap_analysis, report_generation | +| `gap_analysis` | gap_analysis | report_generation | +| `citation_exports` | citation_export | report_generation (via `data` param) | +| `citation_index` | citation_export | report_generation | +| `enhanced_report` | report_generation | pipeline result, API, CLI | + +**Exported in `ResearchPipelineResult.artifacts`** (`pipeline.py:167–182`): +`ranked_papers`, `retrieved_papers`, `paper_analyses`, `paper_clusters`, `synthesis_result`, `gap_analysis`, `citation_exports`, `citation_index`, `enhanced_report`. + +--- + +## Per-stage reference + +### 1. `query_understanding` + +| Field | Value | +|-------|-------| +| Class | `QueryUnderstandingStage` | +| File | `src/research/query_understanding.py` | +| Input (`data`) | `str` — raw query | +| Input artifacts | None | +| Output artifacts | `query_understanding` → `QueryUnderstandingResult` | +| `StageResult` type | `StageResult[QueryUnderstandingResult]` | +| Config keys | None | +| LLM | No — regex/heuristic | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 2. `query_expansion` + +| Field | Value | +|-------|-------| +| Class | `QueryExpansionStage` | +| File | `src/research/query_expansion.py` | +| Input (`data`) | `str \| QueryUnderstandingResult` | +| Input artifacts | None | +| Output artifacts | `expanded_queries` → `ExpandedQuerySet` | +| `StageResult` type | `StageResult[ExpandedQuerySet]` | +| Config keys | `query_expansion.llm_enabled`, `llm_mode`, `max_variants`, `max_sub_questions` | +| LLM | Optional — `AgentRole.EXPANSION` when `query_expansion.llm_enabled`; heuristics always run first | +| Timeout | `pipeline.stage_timeout_seconds` | + +**Note:** `expand_query_llm` uses `get_settings().llm`, not `ctx.config.llm` — potential inconsistency if settings differ. + +--- + +### 3. `retrieval` + +| Field | Value | +|-------|-------| +| Class | `RetrievalStage` | +| File | `src/retrieval/retrieval_stage.py` | +| Input (`data`) | `ExpandedQuerySet` | +| Input artifacts | `cached_papers` (session cache bypass) | +| Output artifacts | `retrieved_papers` → `list[RetrievedPaper]` | +| `StageResult` type | `StageResult[list[RetrievedPaper]]` | +| Config keys | `retrieval.concurrency_limit`, `retrieval.per_provider_limit`, `retrieval.providers.{name}.enabled` | +| LLM | No | +| Timeout | `pipeline.stage_timeout_seconds` | + +**Limit caveat:** Stage always passes `settings.retrieval.per_provider_limit` to `provider.search()` — per-provider `limit` in YAML is ignored. + +--- + +### 4. `deduplication` + +| Field | Value | +|-------|-------| +| Class | `DeduplicationStage` | +| File | `src/retrieval/deduplication.py` | +| Input (`data`) | `list[RetrievedPaper]` | +| Output artifacts | `deduplication_stats` → `dict[str, int]` | +| Config keys | `deduplication.enabled`, `enable_embedding_dedup`, `embedding_similarity_threshold`; `embedding.*` when embedding dedup runs | +| LLM | No — metadata union-find + optional embedding similarity | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 5. `ranking` + +| Field | Value | +|-------|-------| +| Class | `RankingStage` | +| File | `src/research/ranking.py` | +| Input (`data`) | `list[RetrievedPaper]` | +| Output artifacts | `ranked_papers`; `query_embedding`; `paper_embeddings` | +| Config keys | `ranking.top_k`, `ranking.weights.*`, `domain_penalty_multiplier`, `outlier_embedding_gap`, `keyword_collision_max_sim`, `canonical_boost`; `embedding.*` | +| LLM | No | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 6. `relevance_scoring` + +| Field | Value | +|-------|-------| +| Class | `RelevanceScoringStage` | +| File | `src/research/relevance_scoring.py` | +| Input (`data`) | `list[RankedPaper]` | +| Input artifacts | `query_understanding`, `query_embedding`, `paper_embeddings` | +| Output artifacts | `ranked_papers` (filtered); `relevance_filter_reasons` | +| Config keys | `relevance_scoring.*` (min scores, adaptive floor, concept matching) | +| LLM | No | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 7. `clustering` + +| Field | Value | +|-------|-------| +| Class | `ClusteringStage` | +| File | `src/research/clustering.py` | +| Input (`data`) | `list[RankedPaper]` | +| Input artifacts | `paper_embeddings` | +| Output artifacts | `paper_clusters` → `list[PaperCluster]` | +| Config keys | `clustering.*`; `embedding.*`; `ranking.*` (adapter) | +| LLM | No — HDBSCAN + keyword fallback | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 8. `synthesis` + +| Field | Value | +|-------|-------| +| Class | `SynthesisStage` | +| File | `src/analysis/synthesis.py` | +| Input (`data`) | `list[PaperCluster]` | +| Input artifacts | `ranked_papers`; fallback `retrieved_papers` | +| Output artifacts | `paper_extractions`, `paper_analyses`, `synthesis_result`; may refresh `ranked_papers` on recovery | +| Config keys | `synthesis.llm_enabled`, `llm_mode`, `max_llm_papers`, retries, `concurrency`, `circuit_breaker_failures`; `llm.*` | +| LLM | Two-pass when enabled: `AgentRole.EXTRACTION` (per paper, capped) → `AgentRole.SYNTHESIS` (collective). Heuristic fallback when disabled or on failure. | +| Timeout | **`pipeline.synthesis_timeout_seconds`** (600s) | + +**Recovery:** `src/core/stage_recovery.py` — synthesis timeout returns heuristic partial output. + +--- + +### 9. `gap_analysis` + +| Field | Value | +|-------|-------| +| Class | `GapAnalysisStage` | +| File | `src/analysis/gap_analysis.py` | +| Input (`data`) | `SynthesisResult` (may be wrong type; recovered) | +| Input artifacts | `paper_clusters`; `synthesis_result` | +| Output artifacts | `gap_analysis` → `GapAnalysisResult` | +| Config keys | **`synthesis.llm_enabled`** gates LLM (no separate gap flag) | +| LLM | Optional — `AgentRole.GAP_ANALYSIS` when `synthesis.llm_enabled`; else `_heuristic_gap_analysis` | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 10. `citation_export` + +| Field | Value | +|-------|-------| +| Class | `CitationExportStage` | +| File | `src/reporting/citations.py` | +| Input (`data`) | `GapAnalysisResult` (passed through) | +| Input artifacts | `ranked_papers` | +| Output artifacts | `citation_exports`, `citation_index` | +| Config keys | None | +| LLM | No — BibTeX/CSL via `generate_citation_exports` | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +### 11. `report_generation` + +| Field | Value | +|-------|-------| +| Class | `ReportGenerationStage` | +| File | `src/reporting/report_generation.py` | +| Input (`data`) | `dict[str, str]` — citation exports | +| Input artifacts | `synthesis_result`, `gap_analysis`, `paper_clusters`, `paper_analyses`, `ranked_papers`, `citation_index` | +| Output artifacts | `enhanced_report` → `EnhancedResearchReport` | +| Config keys | `relevance_scoring.*` (executive summary embedding floor) | +| LLM | No — deterministic assembly | +| Timeout | `pipeline.stage_timeout_seconds` | + +--- + +## Stage recovery (`src/core/stage_recovery.py`) + +| Stage | Recovery behavior | +|-------|-------------------| +| `synthesis` | Heuristic extraction/synthesis from ranked papers | +| `gap_analysis` | `recover_gap_analysis_output()` — heuristic from synthesis + clusters | +| Others | Return prior `data` unchanged | + +## Data-flow diagram + +```mermaid +flowchart LR + Q[query: str] --> QU[query_understanding] + QU -->|QueryUnderstandingResult| QE[query_expansion] + QE -->|ExpandedQuerySet| RT[retrieval] + RT -->|list RetrievedPaper| DD[deduplication] + DD -->|list RetrievedPaper| RK[ranking] + RK -->|list RankedPaper| RS[relevance_scoring] + RS -->|list RankedPaper| CL[clustering] + CL -->|list PaperCluster| SY[synthesis] + SY -->|SynthesisResult| GA[gap_analysis] + GA -->|GapAnalysisResult| CE[citation_export] + CE -->|dict exports| RG[report_generation] + RG -->|EnhancedResearchReport| OUT[output] +``` + +## Non-pipeline LLM module + +`src/analysis/llm.py` creates a module-level `analysis_agent` at import via `AgentFactory()` — **not** part of the 11-stage pipeline. Used by legacy/orchestrator helper paths. diff --git a/docs/_analysis/config-inventory.md b/docs/_analysis/config-inventory.md new file mode 100644 index 0000000..f1a220c --- /dev/null +++ b/docs/_analysis/config-inventory.md @@ -0,0 +1,268 @@ +# AppSettings / Config Inventory + +Source: `src/config/settings.py`, `config/*.yaml`, `.env.example`, `src/config/resolve_llm_features.py`, `src/config/model_selection.py`. + +## Load precedence + +| Priority | Source | Mechanism | +|----------|--------|-----------| +| 1 (highest) | Constructor kwargs | `AppSettings(retrieval={...})` | +| 2 | Process environment | `RA_` prefix, nested `__` delimiter | +| 3 | `.env` file | Same rules as env (`env_file` in `SettingsConfigDict`) | +| 4 | Merged YAML | `YamlSettingsSource` → `load_yaml_config()` | +| 5 (lowest) | Pydantic field defaults | Nested model defaults in `settings.py` | + +**Post-load resolution:** `resolve_effective_settings()` at pipeline start computes effective `synthesis.llm_enabled`, `query_expansion.llm_enabled`, and Ollama `max_llm_papers` hints. + +**Alternate loader:** `AppSettings.from_yaml()` — constructor > env > YAML > defaults; **skips `.env`**. + +### Doc/code discrepancies + +| Topic | Code | README / `.env.example` | Action for Phase 1 | +|-------|------|-------------------------|-------------------| +| Precedence order | init > env > .env > yaml > defaults | README: env > .env > yaml | Document constructor kwargs as highest | +| `AppSettings` docstring | Omits `.env` step | — | Fix docstring or docs to match runtime | +| `RA_LLM__BASE_URL` default | `http://localhost:11434` (no `/v1`) | `.env.example` uses `/v1` | Document Ollama normalizes via `normalize_openai_base_url()` | +| Debug | `RA_PIPELINE__DEBUG` + `RA_DEBUG` | `.env.example` has `RA_DEBUG=1` active | Warn that example enables debug | +| CrossRef mailto | Optional (fallback User-Agent) | "Required when enabled" | Clarify polite-pool recommendation vs requirement | +| Per-provider limit | `ProviderConfig.limit` exists | Not documented | Document that pipeline uses `per_provider_limit` only | + +--- + +## Top-level `AppSettings` fields + +| Field | Type | Env prefix | Default factory | +|-------|------|------------|-----------------| +| `llm` | `LLMConfig` | `RA_LLM__*` | `LLMConfig()` | +| `embedding` | `EmbeddingConfig` | `RA_EMBEDDING__*` | `EmbeddingConfig()` | +| `ranking` | `RankingConfig` | `RA_RANKING__*` | `RankingConfig()` | +| `query_expansion` | `QueryExpansionConfig` | `RA_QUERY_EXPANSION__*` | `QueryExpansionConfig()` | +| `deduplication` | `DeduplicationConfig` | `RA_DEDUPLICATION__*` | `DeduplicationConfig()` | +| `clustering` | `ClusteringConfig` | `RA_CLUSTERING__*` | `ClusteringConfig()` | +| `relevance_scoring` | `RelevanceScoringConfig` | `RA_RELEVANCE_SCORING__*` | `RelevanceScoringConfig()` | +| `retrieval` | `RetrievalConfig` | `RA_RETRIEVAL__*` | `RetrievalConfig()` | +| `pipeline` | `PipelineConfig` | `RA_PIPELINE__*` | `PipelineConfig()` | +| `memory` | `MemoryConfig` | `RA_MEMORY__*` | `MemoryConfig()` | +| `synthesis` | `SynthesisConfig` | `RA_SYNTHESIS__*` | `SynthesisConfig()` | + +### Non-`AppSettings` env vars + +| Variable | Read in | Purpose | +|----------|---------|---------| +| `RA_CONFIG_DIR` | `settings.py` | Override config directory | +| `RA_DEBUG` | `settings.debug_enabled` | Debug alias (`1`/`true`/`yes`) | +| `S2_API_KEY` | `semantic_scholar.py` | Semantic Scholar API key | +| `RA_CROSSREF_MAILTO` / `CROSSREF_MAILTO` | `crossref.py` | CrossRef polite-pool User-Agent | +| `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `OLLAMA_API_KEY` | `src/models/*` | LLM keys (fallback if `RA_LLM__API_KEY` unset) | + +--- + +## Nested models — fields and defaults + +### `LLMConfig` + +| Field | Default | Env example | +|-------|---------|-------------| +| `provider` | `"ollama"` | `RA_LLM__PROVIDER` | +| `model` | `"auto"` | `RA_LLM__MODEL` | +| `base_url` | `"http://localhost:11434"` | `RA_LLM__BASE_URL` | +| `api_key` | `None` | `RA_LLM__API_KEY` | +| `temperature` | `0.2` | `RA_LLM__TEMPERATURE` | +| `timeout_seconds` | `120` | `RA_LLM__TIMEOUT_SECONDS` | + +**Unused at runtime:** `temperature`, `timeout_seconds` (not passed to pydantic-ai models; stage timeouts are pipeline-level). + +### `EmbeddingConfig` + +| Field | Default | +|-------|---------| +| `model` | `"BAAI/bge-small-en-v1.5"` | +| `batch_size` | `32` | +| `cache_dir` | `"data/embeddings"` | + +### `RankingWeights` + +| Field | Default | +|-------|---------| +| `semantic_relevance` | `0.20` | +| `citation_count` | `0.08` | +| `recency` | `0.08` | +| `venue_quality` | `0.10` | +| `abstract_completeness` | `0.10` | +| `keyword_overlap` | `0.10` | +| `author_prominence` | `0.05` | +| `embedding_similarity` | `0.30` | + +### `RankingConfig` + +| Field | Default | +|-------|---------| +| `top_k` | `25` | +| `weights` | `RankingWeights()` | +| `domain_penalty_multiplier` | `0.5` | +| `outlier_embedding_gap` | `0.12` | +| `keyword_collision_max_sim` | `0.40` | +| `canonical_boost` | `0.0` | + +### `QueryExpansionConfig` + +| Field | Default | Notes | +|-------|---------|-------| +| `llm_mode` | `"auto"` | `on` \| `off` \| `auto` | +| `llm_enabled` | `False` | Resolved at pipeline start | +| `max_variants` | `5` | | +| `max_sub_questions` | `3` | | + +### `DeduplicationConfig` + +| Field | Default | +|-------|---------| +| `enabled` | `True` | +| `enable_embedding_dedup` | `True` | +| `embedding_similarity_threshold` | `0.92` | + +### `ClusteringConfig` + +| Field | Default | +|-------|---------| +| `min_cluster_size` | `2` | +| `min_samples` | `1` | +| `noise_merge_threshold` | `0.5` | +| `max_macro_clusters` | `4` | + +### `RelevanceScoringConfig` + +| Field | Default | +|-------|---------| +| `min_rank_score` | `0.25` | +| `min_embedding_similarity` | `0.35` | +| `require_all_concepts` | `True` | +| `min_papers` | `5` | +| `concept_match_mode` | `"any_group"` | +| `adaptive_embedding` | `True` | +| `keep_percentile` | `25.0` | +| `gap_from_top` | `0.12` | + +### `ProviderConfig` + +| Field | Default | +|-------|---------| +| `enabled` | `True` | +| `limit` | `8` | + +### `RetrievalConfig` + +| Field | Default | +|-------|---------| +| `concurrency_limit` | `4` | +| `per_provider_limit` | `8` | +| `providers` | see table below | + +**Default provider toggles (code):** + +| Provider key | `enabled` | `limit` | +|--------------|-----------|---------| +| `openalex` | `True` | `8` | +| `semantic_scholar` | `True` | `8` | +| `arxiv` | `False` | `8` | +| `crossref` | `False` | `8` | +| `pubmed` | `False` | `8` | +| `core` | `False` | `8` | +| `dblp` | `False` | `8` | + +**Env examples:** +`RA_RETRIEVAL__CONCURRENCY_LIMIT`, `RA_RETRIEVAL__PER_PROVIDER_LIMIT`, +`RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true` + +### `PipelineConfig` + +| Field | Default | +|-------|---------| +| `continue_on_stage_failure` | `True` | +| `stage_timeout_seconds` | `300` | +| `synthesis_timeout_seconds` | `600` | +| `stream_progress` | `True` | +| `debug` | `False` | +| `enabled_stages.*` | all 11 stages `True` | + +### `MemoryConfig` + +| Field | Default | +|-------|---------| +| `db_path` | `"data/research.db"` | +| `cache_enabled` | `False` | + +### `SynthesisConfig` + +| Field | Default | Notes | +|-------|---------|-------| +| `llm_mode` | `"auto"` | | +| `llm_enabled` | `False` | Resolved at pipeline start | +| `max_llm_papers` | `3` | May be overridden by Ollama catalog hints | +| `extraction_max_retries` | `0` | | +| `collective_max_retries` | `0` | | +| `concurrency` | `2` | | +| `circuit_breaker_failures` | `2` | | + +### `debug_enabled` (computed property) + +```python +env_debug = os.environ.get("RA_DEBUG", "").lower() in {"1", "true", "yes"} +return self.pipeline.debug or env_debug +``` + +--- + +## YAML files in `config/` + +| File | Loaded by `AppSettings`? | Merge target | Purpose | +|------|---------------------------|--------------|---------| +| `default.yaml` | Yes (base) | entire settings tree | Full default mirror | +| `models.yaml` | Yes | `llm` section | LLM overrides | +| `ranking.yaml` | Yes | `ranking` section | Ranking overrides | +| `providers.yaml` | Yes | `retrieval` section | Provider toggles | +| `ollama_models.yaml` | **No** | — | Ollama catalog; loaded by `model_selection.py` | +| `canonical_works.yaml` | **No** | — | Ranking boost; loaded by `canonical_works.py` | + +### `ollama_models.yaml` structure + +- `auto_select: bool` +- `fallback: model_name` +- `models[]`: `name`, `label`, `min_ram_gb`, `recommended_ram_gb`, `disk_gb`, `priority`, `synthesis.llm_enabled`, `synthesis.max_llm_papers` + +### `canonical_works.yaml` structure + +- `works[]`: `title`, `authors[]`, `year`, `doi_prefix` + +--- + +## `.env.example` inventory + +| Variable | In example | Notes | +|----------|------------|-------| +| `S2_API_KEY` | commented | Semantic Scholar | +| `RA_CROSSREF_MAILTO` / `CROSSREF_MAILTO` | commented | CrossRef | +| `RA_LLM__PROVIDER` | `ollama` | | +| `RA_LLM__MODEL` | `auto` | | +| `RA_LLM__BASE_URL` | `http://localhost:11434/v1` | | +| `RA_LLM__API_KEY` | `ollama` | | +| `RA_SYNTHESIS__LLM_ENABLED` | commented | | +| OpenAI / Anthropic blocks | commented | | +| `RA_RANKING__TOP_K` | `25` | | +| `RA_PIPELINE__DEBUG` | `false` | | +| `RA_PIPELINE__STREAM_PROGRESS` | `true` | | +| `RA_DEBUG` | **`1` (active)** | Enables debug despite `pipeline.debug=false` | +| `RA_CONFIG_DIR` | commented | | + +--- + +## LLM feature resolution env vars + +| Env var | Effect | +|---------|--------| +| `RA_SYNTHESIS__LLM_ENABLED` | Force synthesis LLM on/off (overrides `llm_mode`) | +| `RA_QUERY_EXPANSION__LLM_ENABLED` | Force query expansion LLM on/off | +| `RA_SYNTHESIS__LLM_MODE` | `auto` \| `on` \| `off` | +| `RA_QUERY_EXPANSION__LLM_MODE` | `auto` \| `on` \| `off` | + +See [llm-resolution-tree.md](llm-resolution-tree.md) for full decision tree. diff --git a/docs/_analysis/llm-resolution-tree.md b/docs/_analysis/llm-resolution-tree.md new file mode 100644 index 0000000..4f84e88 --- /dev/null +++ b/docs/_analysis/llm-resolution-tree.md @@ -0,0 +1,228 @@ +# LLM Resolution Tree + +Source: `src/config/settings.py`, `src/config/resolve_llm_features.py`, `src/config/model_selection.py`, `src/models/factory.py`, `src/models/{ollama,openai,anthropic}.py`, `src/models/base.py`. + +## Overview + +LLM resolution happens in two phases: + +1. **Settings load** — `AppSettings` from kwargs/env/.env/YAML/defaults +2. **Pipeline start** — `resolve_effective_settings()` computes feature flags and Ollama hints +3. **Agent creation** — `AgentFactory` resolves `model: auto` → concrete name, then instantiates provider + +Embedding model (`embedding.model`) is **separate** from LLM — used by deduplication, ranking, relevance_scoring, clustering. + +--- + +## Phase 1: Settings load + +See [config-inventory.md](config-inventory.md) for full field list. + +**Key LLM fields:** + +| Field | Default | Env | +|-------|---------|-----| +| `llm.provider` | `"ollama"` | `RA_LLM__PROVIDER` | +| `llm.model` | `"auto"` | `RA_LLM__MODEL` | +| `llm.base_url` | `"http://localhost:11434"` | `RA_LLM__BASE_URL` | +| `llm.api_key` | `None` | `RA_LLM__API_KEY` | +| `synthesis.llm_mode` | `"auto"` | `RA_SYNTHESIS__LLM_MODE` | +| `query_expansion.llm_mode` | `"auto"` | `RA_QUERY_EXPANSION__LLM_MODE` | + +--- + +## Phase 2: Feature flag resolution (`resolve_effective_settings`) + +Called from `ResearchPipeline.execute()` before stages run. + +### Precedence per feature (synthesis, query_expansion) + +1. `RA_{SECTION}__LLM_ENABLED` env override (`true`/`false`/`1`/`0`/etc.) +2. `llm_mode: on` → enabled; `llm_mode: off` → disabled +3. `llm_mode: auto` → rules below + +### Auto-mode rules (`_resolve_auto_llm_enabled`) + +| Provider | synthesis LLM | query_expansion LLM | +|----------|---------------|---------------------| +| `openai`, `anthropic` | **Always enabled** | **Always enabled** | +| `ollama` | Enabled if `ollama_models.yaml` entry has `synthesis.llm_enabled: true` for resolved model | Same catalog hint | +| Other | Disabled | Disabled | + +### Ollama synthesis hints + +When provider is `ollama` and model resolves from catalog, `max_llm_papers` may be updated from catalog entry (e.g. 8B model → 5 papers). + +```mermaid +flowchart TD + START[resolve_effective_settings settings] --> PROVIDER[provider = llm.provider.lower] + PROVIDER --> MODEL{llm.model is auto/empty?} + MODEL -->|yes| RESOLVE[resolve_llm_model_name from ollama_models.yaml + RAM/disk] + MODEL -->|no| FIXED[use llm.model as-is] + RESOLVE --> SYN + FIXED --> SYN + + SYN[_resolve_llm_enabled section=synthesis] + SYN --> SYN_ENV{RA_SYNTHESIS__LLM_ENABLED set?} + SYN_ENV -->|true/false| SYN_MODE[force on/off] + SYN_ENV -->|unset| SYN_CFG{synthesis.llm_mode} + SYN_CFG -->|on| SYN_ON[llm_enabled=true] + SYN_CFG -->|off| SYN_OFF[llm_enabled=false] + SYN_CFG -->|auto| SYN_AUTO{provider} + SYN_AUTO -->|openai/anthropic| SYN_CLOUD[true] + SYN_AUTO -->|ollama| SYN_HINT[synthesis_hints_for_model] + SYN_AUTO -->|other| SYN_FALSE[false] + + SYN_ON --> EXP + SYN_OFF --> EXP + SYN_CLOUD --> EXP + SYN_HINT --> EXP + SYN_FALSE --> EXP + + EXP[_resolve_llm_enabled section=query_expansion] + EXP --> EXP_SAME[same precedence as synthesis] + EXP_SAME --> OLLAMA{provider == ollama?} + OLLAMA -->|yes| HINTS[apply max_llm_papers from catalog hints] + OLLAMA -->|no| DONE[return settings copy with resolved flags] + HINTS --> DONE +``` + +--- + +## Phase 3: Model name resolution (`model_selection.py`) + +When `llm.model` is `"auto"` or empty: + +```mermaid +flowchart TD + REQ[resolve_llm_model_name] --> RTM[resolve_target_model] + RTM --> SRC{model source priority} + SRC -->|CLI arg| EXPLICIT + SRC -->|RA_LLM__MODEL env| ENV + SRC -->|YAML llm.model| YAML + SRC -->|auto| AUTO + + AUTO --> CATALOG[load config/ollama_models.yaml] + CATALOG --> AUTOSEL{catalog.auto_select?} + AUTOSEL -->|false| FALLBACK[catalog.fallback model] + AUTOSEL -->|true| RES[detect_system_resources RAM/disk/swap] + RES --> PICK[highest priority model that fits resources] + PICK -->|none fit| FALLBACK + + EXPLICIT --> NAME[concrete model_name string] + ENV --> NAME + YAML --> NAME + FALLBACK --> NAME + PICK --> NAME +``` + +**Resource detection:** RAM, disk, swap pressure can downgrade model selection (see `tests/test_model_selection.py`). + +--- + +## Phase 4: Provider and agent creation (`factory.py`) + +```mermaid +flowchart TD + AF[AgentFactory config] --> RESOLVE[_resolve_config] + RESOLVE --> AUTO{model == auto?} + AUTO -->|yes| RESNAME[resolve_llm_model_name] + AUTO -->|no| CFG[use config as-is] + RESNAME --> CREATE[create_llm_provider] + CFG --> CREATE + CREATE --> LOOKUP[get_llm_provider_class by provider name] + LOOKUP --> MODEL[provider.create_model config] + MODEL --> AGENT[Agent model + ROLE_SYSTEM_PROMPTS role] +``` + +### LLM provider registry + +| Provider key | Class | File | Model backend | +|--------------|-------|------|---------------| +| `ollama` | `OllamaProvider` | `src/models/ollama.py` | OpenAI-compatible API at normalized `base_url` | +| `openai` | `OpenAIProviderImpl` | `src/models/openai.py` | pydantic-ai OpenAI model | +| `anthropic` | `AnthropicProviderImpl` | `src/models/anthropic.py` | pydantic-ai Anthropic model | + +### API key resolution + +| Provider | Key sources (priority) | +|----------|------------------------| +| Ollama | `RA_LLM__API_KEY` → `OLLAMA_API_KEY` → default `"ollama"` | +| OpenAI | `RA_LLM__API_KEY` → `OPENAI_API_KEY` (required) | +| Anthropic | `RA_LLM__API_KEY` → `ANTHROPIC_API_KEY` (required) | + +### Base URL normalization (Ollama / OpenAI-compatible) + +`normalize_openai_base_url()` appends `/v1` if missing — explains discrepancy between code default (`http://localhost:11434`) and `.env.example` (`.../v1`). + +--- + +## Agent roles in pipeline + +| Role | Enum | Used in stage | Purpose | +|------|------|---------------|---------| +| `EXPANSION` | `AgentRole.EXPANSION` | query_expansion | JSON variants + sub_questions | +| `EXTRACTION` | `AgentRole.EXTRACTION` | synthesis Pass A | Per-paper structured extraction | +| `SYNTHESIS` | `AgentRole.SYNTHESIS` | synthesis Pass B | Cross-paper synthesis JSON | +| `GAP_ANALYSIS` | `AgentRole.GAP_ANALYSIS` | gap_analysis | Gaps/opportunities JSON | +| `ANALYSIS` | `AgentRole.ANALYSIS` | `src/analysis/llm.py` only | Legacy module-level agent | + +--- + +## Runtime LLM call sites (after resolution) + +```mermaid +flowchart TD + RESOLVED[ctx.config after resolve_effective_settings] + + RESOLVED --> QE{query_expansion.llm_enabled?} + QE -->|yes| QE_AGENT[AgentFactory EXPANSION stream_agent_text] + QE -->|no| QE_SKIP[heuristics only] + + RESOLVED --> SY{synthesis.llm_enabled?} + SY -->|yes| SY_A[AgentFactory EXTRACTION up to max_llm_papers] + SY_A --> SY_B[AgentFactory SYNTHESIS collective] + SY -->|no| SY_H[heuristic extraction + synthesis] + + RESOLVED --> GA{synthesis.llm_enabled?} + GA -->|yes| GA_AGENT[AgentFactory GAP_ANALYSIS structured response] + GA -->|no| GA_H[heuristic from synthesis fields] +``` + +**Important coupling:** Gap analysis LLM is gated by **`synthesis.llm_enabled`**, not a separate `gap_analysis.llm_mode`. + +--- + +## Test coverage (`tests/test_resolve_llm_features.py`) + +| Test | Confirms | +|------|----------| +| `test_llm_mode_auto_8b` | Ollama 8B + auto → both features enabled, `max_llm_papers=5` | +| `test_llm_mode_auto_3b` | Ollama 3B + auto → both features disabled | +| `test_llm_mode_on_off` | Explicit on/off overrides catalog | +| `test_env_llm_enabled_overrides_mode` | `RA_SYNTHESIS__LLM_ENABLED` overrides `llm_mode: off` | +| `test_cloud_provider_auto_enables_llm` | OpenAI + auto → both enabled | +| `test_auto_resolves_model_name` | End-to-end YAML + auto resolution | + +--- + +## Known gaps (document in Phase 1) + +| Issue | Detail | +|-------|--------| +| `llm.timeout_seconds` | Configured but unused; stage timeouts are pipeline-level | +| `llm.temperature` | Not passed to pydantic-ai model constructors | +| `expand_query_llm` | Uses `get_settings().llm` instead of `ctx.config.llm` | +| `analysis/llm.py` | Global agent at import — separate from pipeline | +| Default quality | `synthesis.llm_enabled=false` + `query_expansion.llm_enabled=false` → heuristic reports | + +## Default quality implications + +With Ollama `llama3.2:3b` (fallback) and `llm_mode: auto`: + +- Synthesis LLM: **off** +- Query expansion LLM: **off** +- Gap analysis: **heuristic** (derived from synthesis heuristics) +- Report placeholders: *"Details inferred from abstract only"*, etc. + +See `docs/quality/known-issues.md` for user-facing quality analysis. diff --git a/docs/_analysis/provider-http-matrix.md b/docs/_analysis/provider-http-matrix.md new file mode 100644 index 0000000..684e667 --- /dev/null +++ b/docs/_analysis/provider-http-matrix.md @@ -0,0 +1,171 @@ +# Provider HTTP Matrix — Retrieval + +Source: `src/retrieval/providers/`, `src/retrieval/providers/registry.py`, `src/retrieval/retrieval_stage.py`, `src/retrieval/orchestrator.py`. + +**HTTP client:** `aiohttp` (session created in `RetrievalStage` or passed to health checks). + +## Provider summary + +| Key | File | Status | Base URL / endpoint | Auth | Retries / timeouts | YAML enable key | +|-----|------|--------|---------------------|------|-------------------|-----------------| +| `openalex` | `openalex.py` | **Live** | `GET https://api.openalex.org/works?search=&per-page=` | None | 3 retries, exp backoff; search **60s**, health **15s** | `retrieval.providers.openalex.enabled` (default `true`) | +| `semantic_scholar` | `semantic_scholar.py` | **Live** | `GET https://api.semanticscholar.org/graph/v1/paper/search/bulk?query=&limit=&fields=...` | `S2_API_KEY` → `x-api-key` header (optional) | 3 retries; **429** → sleep `Retry-After` (default 60s); search **60s**, health **15s** | `retrieval.providers.semantic_scholar.enabled` (default `true`) | +| `arxiv` | `arxiv.py` | **Live** | `GET https://export.arxiv.org/api/query?search_query=&start=0&max_results=` | None | 3 retries; search **60s**, health **15s** | `retrieval.providers.arxiv.enabled` (default `false`) | +| `crossref` | `crossref.py` | **Live** | `GET https://api.crossref.org/works?query=&rows=` | `RA_CROSSREF_MAILTO` or `CROSSREF_MAILTO` → `User-Agent: ResearchAssistant/1.0 (mailto:…)` | 3 retries; **429** → `Retry-After`; search **60s**, health **15s** | `retrieval.providers.crossref.enabled` (default `false`) | +| `pubmed` | `pubmed.py` | **Stub** | Planned NCBI E-utilities | — | `NotImplementedError` on `search()` | `retrieval.providers.pubmed.enabled` (default `false`) | +| `core` | `core_provider.py` | **Stub** | Planned CORE API v3 | — | `NotImplementedError` on `search()` | `retrieval.providers.core.enabled` (default `false`) | +| `dblp` | `dblp.py` | **Stub** | Planned DBLP JSON API | — | `NotImplementedError` on `search()` | `retrieval.providers.dblp.enabled` (default `false`) | + +**Stub base:** `src/retrieval/providers/_stub_base.py` — `search()` raises `NotImplementedError`; `health_check()` returns unhealthy. + +## Registry and selection + +``` +get_enabled_providers(settings) + → iterate settings.retrieval.providers + → skip if enabled=false or name not in _PROVIDER_CLASSES + → create_provider(name, config) +``` + +File: `src/retrieval/providers/registry.py:62–76` + +**Extensibility:** `register_provider()` adds to `_PROVIDER_CLASSES` (tested in `test_phase3_extensibility.py`). + +## Per-provider details + +### OpenAlex (`openalex`) + +| Attribute | Value | +|-----------|-------| +| Search URL | `https://api.openalex.org/works` | +| Query params | `search`, `per-page` | +| Auth | None | +| Rate limiting | Retry on failure; no explicit 429 handler | +| Normalization | Maps OpenAlex work JSON → `RetrievedPaper` | +| Health check | Lightweight query against works endpoint, 15s timeout | + +### Semantic Scholar (`semantic_scholar`) + +| Attribute | Value | +|-----------|-------| +| Search URL | `https://api.semanticscholar.org/graph/v1/paper/search/bulk` | +| Fields requested | `title,abstract,year,venue,url,externalIds` | +| Auth | `S2_API_KEY` env → optional `x-api-key` header | +| Rate limiting | 429 handling with `Retry-After` header | +| Health check | Minimal search, 15s timeout | + +### arXiv (`arxiv`) + +| Attribute | Value | +|-----------|-------| +| Search URL | `https://export.arxiv.org/api/query` | +| Query format | Preserves arXiv field syntax (e.g. `ti:`, `abs:`) | +| Auth | None | +| Health check | `max_results=1`, 15s timeout | + +### CrossRef (`crossref`) + +| Attribute | Value | +|-----------|-------| +| Search URL | `https://api.crossref.org/works` | +| Query params | `query`, `rows` | +| Auth | Polite pool via mailto in User-Agent (optional but recommended) | +| Rate limiting | 429 handling with `Retry-After` | +| Health check | `rows=1`, 15s timeout | + +### Stubs (PubMed, CORE, DBLP) + +| Attribute | Value | +|-----------|-------| +| Registration | Present in `_PROVIDER_CLASSES`; default `enabled=false` | +| `search()` | Raises `NotImplementedError` | +| `health_check()` | Returns `healthy=False` with stub message | +| Normalization | Field maps defined for future implementation | +| Risk | Enabling in YAML → caught by `_safe_search` → warning + empty results per provider | + +## Retrieval stage behavior + +File: `src/retrieval/retrieval_stage.py` + +| Behavior | Detail | +|----------|--------| +| Provider limit | Always uses `settings.retrieval.per_provider_limit` — **ignores** `ProviderConfig.limit` | +| Concurrency | `settings.retrieval.concurrency_limit` caps parallel searches across expanded query variants | +| Per-query providers | All enabled providers searched in parallel via `asyncio.gather` | +| Failure handling | `_safe_search` catches exceptions → warning, empty list for that provider | +| Partial flag | Stage marked partial when any provider fails | +| Cache bypass | `cached_papers` initial artifact skips network if session cache hit | + +## CLI vs full pipeline + +| Aspect | Full pipeline (`run_research` / API) | CLI helper (`run_research_helper`) | +|--------|--------------------------------------|-------------------------------------| +| File | `orchestrator.py:110–187` | `orchestrator.py:190–276` | +| Settings | Full `AppSettings()` merge | Constructor override replaces `providers` dict | +| Providers | All `enabled=true` in config | **Hardcoded:** OpenAlex + Semantic Scholar only | +| Limit | `per_provider_limit` (default 8) | `k_each` param → `per_provider_limit` + both provider limits | +| YAML toggles | Honored (e.g. enable arXiv) | **Ignored** for provider set | +| Entry points | API, interactive session, programmatic | `python -m src "query"` | + +```python +# orchestrator.py:207–215 — CLI helper override +settings = AppSettings( + retrieval={ + "per_provider_limit": k_each, + "providers": { + "openalex": {"enabled": True, "limit": k_each}, + "semantic_scholar": {"enabled": True, "limit": k_each}, + }, + } +) +``` + +## Config key quick reference + +| Config key | Default | Used by pipeline? | +|------------|---------|-------------------| +| `retrieval.concurrency_limit` | `4` | Yes | +| `retrieval.per_provider_limit` | `8` | Yes (actual search limit) | +| `retrieval.providers..enabled` | see defaults | Yes (full pipeline only) | +| `retrieval.providers..limit` | `8` | **No** (overridden by `per_provider_limit`) | + +## Env override examples + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +RA_RETRIEVAL__PER_PROVIDER_LIMIT=10 +RA_RETRIEVAL__CONCURRENCY_LIMIT=6 +S2_API_KEY=your_key +RA_CROSSREF_MAILTO=you@example.com +``` + +## Graceful fallback flow + +```mermaid +flowchart TD + Q[Expanded query variant] --> GATHER[asyncio.gather all enabled providers] + GATHER --> P1[openalex.search] + GATHER --> P2[semantic_scholar.search] + GATHER --> P3[other providers...] + P1 -->|success| MERGE[Merge papers] + P1 -->|exception| W1[Warning + empty list] + P2 -->|success| MERGE + P2 -->|exception| W2[Warning + empty list] + W1 --> MERGE + W2 --> MERGE + MERGE --> DEDUP[deduplication stage] +``` + +## Documentation admonitions (Phase 1) + +```markdown +!!! warning "Stub provider" + PubMed is not yet implemented. Enabling it raises `NotImplementedError` + (caught gracefully — empty results for that provider). + +!!! info "CLI vs full pipeline" + `python -m src "query"` uses OpenAlex + Semantic Scholar only via + `run_research_helper`. Enable additional providers via config when using + the API or programmatic `build_pipeline()`. +``` diff --git a/docs/_analysis/test-behavior-index.md b/docs/_analysis/test-behavior-index.md new file mode 100644 index 0000000..55d1738 --- /dev/null +++ b/docs/_analysis/test-behavior-index.md @@ -0,0 +1,504 @@ +# Test Behavior Index + +Source: all 28 files matching `tests/test_*.py`. Internal reference for `docs/development/testing.md`. + +## Summary by domain + +| Domain | Test files | +|--------|------------| +| Config / LLM resolution | `test_config_settings.py`, `test_resolve_llm_features.py`, `test_model_selection.py` | +| Pipeline core | `test_pipeline_core.py`, `test_paper_adapters.py`, `test_phase3_extensibility.py` | +| Research stages | `test_research_stages.py`, `test_retrieval_stage.py`, `test_synthesis.py`, `test_research_quality.py` | +| Retrieval providers | `test_providers.py` | +| Embeddings | `test_embeddings.py` | +| Reporting / export | `test_reporting.py`, `test_export.py` | +| LLM layer | `test_llm_providers.py`, `test_graceful_response_handling.py` | +| Orchestrator / degradation | `test_json_parsing_bug_exploration.py`, `test_json_parsing_preservation.py` | +| CLI / interactive | `test_main_mode_detection.py`, `test_interactive_mode.py`, `test_complete_workflow.py`, `test_input_handler.py`, `test_message_formatting.py`, `test_signal_handling.py`, `test_interactive_filters.py` | +| Memory | `test_memory.py` | +| Models | `test_models.py` | +| Progress | `test_progress_reporter.py` | + +--- + +## Per-file index + +### `test_config_settings.py` + +| | | +|---|---| +| **Modules** | `src.config.settings` | +| **Fixtures** | `temp_config_dir` — tmp YAML overlays | +| **Mocks** | `monkeypatch` for `RA_*` env | + +| Test | Behavior | +|------|----------| +| `test_merges_default_and_overlay_files` | YAML merge: overlay wins | +| `test_loads_from_yaml` | `AppSettings.from_yaml` | +| `test_env_overrides_yaml` | `RA_LLM__MODEL`, `RA_RANKING__TOP_K` override YAML | +| `test_constructor_overrides_env_and_yaml` | Init kwargs beat env | +| `test_debug_enabled_from_env` | `RA_DEBUG=1` → `debug_enabled` | +| `test_retrieval_providers_defaults` | OpenAlex enabled by default | + +--- + +### `test_resolve_llm_features.py` + +| | | +|---|---| +| **Modules** | `src.config.resolve_llm_features` | +| **Fixtures** | `catalog_dir` — Ollama catalog with 8B/3B models | + +| Test | Behavior | +|------|----------| +| `test_llm_mode_auto_8b` | Auto + 8B → LLM on, `max_llm_papers=5` | +| `test_llm_mode_auto_3b` | Auto + 3B → LLM off | +| `test_llm_mode_on_off` | Explicit on/off overrides catalog | +| `test_env_llm_enabled_overrides_mode` | Env bool overrides `llm_mode: off` | +| `test_cloud_provider_auto_enables_llm` | OpenAI auto-enables LLM | +| `test_auto_resolves_model_name` | End-to-end auto resolution | + +--- + +### `test_model_selection.py` + +| | | +|---|---| +| **Modules** | `src.config.model_selection` | +| **Fixtures** | `catalog_dir` | + +| Class | Behavior | +|-------|----------| +| `TestModelCatalog` | Catalog load, case-insensitive lookup | +| `TestModelSelection` | RAM-based selection, fallback, swap pressure, explicit/env resolution | +| `TestOllamaListParsing` | `ollama list` stdout parsing | + +--- + +### `test_embeddings.py` + +| | | +|---|---| +| **Modules** | `src.embeddings.cache`, `src.embeddings.sentence_transformers` | +| **Mocks** | `patch.object(provider, "_load_model")`, `MagicMock` encode | + +| Class | Behavior | +|-------|----------| +| `TestEmbeddingCache` | Disk cache round-trip, model-specific keys, batch hit/miss | +| `TestSentenceTransformerEmbeddingProvider` | Cache reuse, cosine similarity, empty input shape | + +--- + +### `test_models.py` + +| | | +|---|---| +| **Modules** | `src.retrieval.models` | + +| Class | Behavior | +|-------|----------| +| `TestRetrievedPaper` | Legacy `source`/`provider`/`paper_id` aliases | +| `TestPipelineModels` | Pydantic defaults for pipeline models | +| `TestEnhancedResearchReport` | `to_research_report` adapter | + +--- + +### `test_pipeline_core.py` + +| | | +|---|---| +| **Modules** | `src.core.pipeline`, `src.core.context`, `src.core.metrics` | +| **Fixtures** | `EchoStage`, `FailingStage`, `PartialStage` stubs | + +| Test | Behavior | +|------|----------| +| `test_pipeline_runs_stages_in_order` | Single stage passes query | +| `test_pipeline_continues_after_stage_failure` | `continue_on_stage_failure=True` | +| `test_pipeline_skips_disabled_stages` | `enabled_stages` gate | +| `test_pipeline_records_partial_stage` | Stage `partial=True` → warnings | +| `test_pipeline_isolates_stages_and_preserves_order` | A→B→C chaining | +| `test_pipeline_propagates_partial_from_multiple_stages` | Multiple partial warnings | +| `test_pipeline_metrics_emit_and_finalize` | Metrics for stages/retrieval/ranking/clustering/LLM | + +--- + +### `test_paper_adapters.py` + +| | | +|---|---| +| **Modules** | `src.core.paper_adapters` | + +| Test | Behavior | +|------|----------| +| `test_ensure_ranked_papers_from_retrieved` | Retrieved → ranked conversion + warnings | +| `test_ensure_ranked_papers_passthrough` | Already-ranked unchanged | + +--- + +### `test_research_stages.py` + +| | | +|---|---| +| **Modules** | query_expansion, ranking, relevance_scoring, clustering, deduplication, pipeline | +| **Fixtures** | `MockEmbeddingProvider`, `FixedEmbeddingProvider`, `_paper()` | +| **Mocks** | `monkeypatch` HDBSCAN; `RetrievalStub` | + +| Class | Scope | +|-------|-------| +| `TestQueryExpansion` | Heuristic expansion, stability, metrics | +| `TestRanking` | Ordering, determinism, penalties, canonical boost, embeddings | +| `TestRelevanceScoring` | Cached embeddings, `min_papers` floor, adaptive floor | +| `TestClustering` | HDBSCAN groups, noise→macro merge, singleton handling | +| `TestDeduplication` | DOI dedup, embedding near-dup, richer duplicate kept | +| `TestMetadataSanity` | Year/DOI correction, future year flags | +| `TestResearchPipelineStages` | End-to-end stub pipeline through clustering | + +**Edge cases:** Homonym decoy papers; adaptive embedding floor; all-noise HDBSCAN → macro themes. + +--- + +### `test_research_quality.py` + +| | | +|---|---| +| **Modules** | query_expansion, ranking, relevance_scoring, clustering, report_generation | +| **Fixtures** | `DOMAIN_CASES` (NLP, biomedical, climate, economics); `FixedEmbeddingProvider` | + +| Class | Behavior | +|-------|----------| +| `TestMultiDomainExpansion` | No degenerate variants across 4 domains | +| `TestMultiDomainRanking` | Embedding outlier demotes homonym decoys | +| `TestMultiDomainRelevance` | Adaptive filter drops decoy | +| `TestMultiDomainExecutiveSummary` | Summary excludes decoy terms | +| `TestNoDomainSpecificBranches` | No ML-specific branch constants in source | +| `TestMultiDomainClustering` | Noise merge → 1–4 macro themes | + +--- + +### `test_retrieval_stage.py` + +| | | +|---|---| +| **Modules** | `src.retrieval.retrieval_stage`, `src.core.pipeline` | +| **Fixtures** | `SuccessProvider`, `FailingProvider`, `EmptyProvider` | +| **Mocks** | `patch get_enabled_providers` | + +| Test | Behavior | +|------|----------| +| `test_search_query_continues_when_one_provider_fails` | Partial success + warning | +| `test_search_query_returns_empty_when_all_providers_fail` | Empty + 2 warnings | +| `test_retrieve_papers_merges_successful_provider_results` | Multi-provider merge | +| `test_retrieval_stage_marks_partial_when_provider_fails` | `partial=True`, metrics | +| `test_retrieval_stage_not_partial_when_all_providers_succeed` | Full success | +| `test_pipeline_continues_after_retrieval_provider_failure` | Downstream runs after partial retrieval | + +--- + +### `test_providers.py` + +| | | +|---|---| +| **Modules** | `src.retrieval.providers` (OpenAlex, S2, arXiv, CrossRef) | +| **Mocks** | `patch` registry; `AsyncMock` aiohttp for health checks | + +| Class | Behavior | +|-------|----------| +| `TestProviderRegistry` | Builtins registered; enabled/disabled filtering; `register_provider` | +| `test_search_enabled_providers_falls_back_when_one_provider_fails` | Cross-provider fallback | +| `test_health_check_enabled_providers_reports_status` | Healthy vs unhealthy | +| `test_arxiv/crossref_health_check_uses_lightweight_query` | Minimal health queries | +| `TestProviderNormalization` | Field mapping for all 4 live providers | + +--- + +### `test_synthesis.py` + +| | | +|---|---| +| **Modules** | `src.analysis.synthesis`, `src.analysis.gap_analysis`, `src.core.stage_recovery` | +| **Mocks** | `MagicMock(EnhancedResponseHandler)`; `patch create_llm_agent` | + +| Class | Behavior | +|-------|----------| +| `TestSynthesisModels` | Pydantic schema validation | +| `TestHeuristicSynthesis` | Abstract extraction, aggregation, conflict detection | +| `TestSynthesisWorkflow` | LLM two-pass; heuristic fallback | +| `TestGapAnalysis` | LLM + heuristic gap analysis | +| `TestSynthesisPipelineStages` | Stage artifacts; mini-pipeline synthesis→gap | +| `TestStageRecovery` | Timeout recovery; `max_llm_papers` cap; skip LLM when disabled | + +--- + +### `test_reporting.py` + +| | | +|---|---| +| **Modules** | `src.reporting.*`, `src.retrieval.orchestrator` | +| **Mocks** | `patch run_research_with_result` | + +| Class | Behavior | +|-------|----------| +| `TestMarkdownReporting` | Enhanced/legacy markdown, JSON/HTML, partial notice | +| `TestCitationExport` | Citation keys, BibTeX | +| `TestExecutiveSummary` | Template summary; embedding-gated findings | +| `TestReportGenerationStage` | Report assembly from artifacts | +| `TestOrchestratorFacade` | CLI prints enhanced report / no-results | + +--- + +### `test_export.py` + +| | | +|---|---| +| **Modules** | `src.export.{bibtex,apa,mla,chicago}` | + +| Class | Behavior | +|-------|----------| +| `TestBibTeXExport`, `TestAPAExport`, `TestMLAExport`, `TestChicagoExport` | Style-specific formatting | +| `TestUnifiedExport` | `generate_citation_exports` aggregator | + +--- + +### `test_llm_providers.py` + +| | | +|---|---| +| **Modules** | `src.models.{ollama,openai,anthropic,factory}` | +| **Mocks** | `patch` OpenAI/Pydantic AI; `patch.dict` for missing keys | + +| Class | Behavior | +|-------|----------| +| `TestNormalizeOpenAIBaseUrl` | Appends `/v1` | +| `TestLLMProviderRegistry` | Default Ollama; unknown → `KeyError` | +| `TestOllamaProvider`, `TestOpenAIProvider` | Endpoint wiring; API key required | +| `TestAgentFactory` | Role prompts; `create_llm_agent` compat | + +--- + +### `test_memory.py` + +| | | +|---|---| +| **Modules** | `src.memory.store` | +| **Fixtures** | `memory_store` (tmp SQLite) | + +| Test | Behavior | +|------|----------| +| `test_create_and_load_session` | Session CRUD | +| `test_save_search_papers_and_report` | Search cache, papers, reports | +| `test_cache_key_is_stable` | Case-insensitive cache key | +| `test_update_session_context` | JSON context persistence | + +--- + +### `test_progress_reporter.py` + +| | | +|---|---| +| **Modules** | `src.utils.progress_reporter` | +| **Mocks** | `patch sys.stderr.isatty` | + +| Test | Behavior | +|------|----------| +| `TestStageLabels` | Friendly labels | +| `TestProgressReporter` | Disabled → blocking; non-TTY disables reporter | + +--- + +### `test_graceful_response_handling.py` + +| | | +|---|---| +| **Modules** | `src.utils.{response_models,retry_manager,quality_monitor,enhanced_validation,content_quality,json_processing,model_adaptation,fallback_processing}` | + +| Class | Behavior | +|-------|----------| +| `TestRetryManager` | Retry rules, prompt enhancement | +| `TestQualityMonitor` | Success/failure recording | +| `TestEnhancedValidation` | Retry strategy mapping | +| `TestContentQuality` | Empty/insufficient/incomplete analysis detection | +| `TestQueryAnalyzer` | Query broadening suggestions | +| `TestRelevanceScorer` | Paper relevance ordering | +| `TestJSONProcessing` | Extract, parse errors, validation | +| `TestModelAdaptation` | GPT/Claude detection, markdown stripping | +| `TestFallbackProcessing` | Unstructured text → structured fallback | + +**Note:** Does not exercise `EnhancedResponseHandler` end-to-end. + +--- + +### `test_json_parsing_bug_exploration.py` + +| | | +|---|---| +| **Modules** | `src.retrieval.orchestrator.run_research_helper` | +| **Fixtures** | `tests.helpers.pipeline_mocks.mock_pipeline_result` | + +| Class | Behavior | +|-------|----------| +| `TestJSONParsingBugCondition` | Malformed JSON still prints partial report | +| `TestScopedBugConditionProperty` | 12 parametrized malformed JSON cases | + +--- + +### `test_json_parsing_preservation.py` + +| | | +|---|---| +| **Modules** | `src.retrieval.orchestrator`, `src.utils.message_formatter` | + +| Class | Behavior | +|-------|----------| +| `TestJSONParsingPreservation` | Valid JSON, code blocks, network/no-results messages | +| `TestPreservationProperties` | Parametrized valid JSON shapes | + +--- + +### `test_main_mode_detection.py` + +| | | +|---|---| +| **Modules** | `src.__main__.main` | +| **Mocks** | `patch sys.argv`, `asyncio.run`, helper functions | + +| Class | Behavior | +|-------|----------| +| `TestMainModeDetection` | Query arg → batch; no arg → interactive | +| `TestModeDetectionEdgeCases` | `""` → interactive; `" "` → batch | + +--- + +### `test_interactive_mode.py` + +| | | +|---|---| +| **Modules** | `src.__main__.run_interactive_mode` | +| **Mocks** | Session helpers, `get_user_query`, `builtins.input` | + +| Test | Behavior | +|------|----------| +| Welcome/farewell | UX copy | +| Query loop | Separators, multi-query, KeyboardInterrupt | +| Integration | Full session flows | + +--- + +### `test_complete_workflow.py` + +| | | +|---|---| +| **Modules** | `src.__main__`, `src.utils.input_handler` | +| **Marks** | `@pytest.mark.slow` subprocess tests | + +| Class | Behavior | +|-------|----------| +| `TestCompleteInteractiveWorkflow` | End-to-end interactive flows | +| `TestSystemLevelIntegration` | Real `python -m src` subprocess | + +--- + +### `test_input_handler.py` + +| | | +|---|---| +| **Modules** | `src.utils.input_handler` | + +| Test | Behavior | +|------|----------| +| Empty/whitespace reprompt | Error + retry | +| exit/quit | Returns `None` | +| EOF | Returns `None` | + +--- + +### `test_message_formatting.py` + +| | | +|---|---| +| **Modules** | `src.utils.message_formatter` | + +| Test | Behavior | +|------|----------| +| Prompt consistency | `query_prompt()` | +| Error prefix | `❌ Error:` | +| Separators | `-` * 60 between results | + +--- + +### `test_signal_handling.py` + +| | | +|---|---| +| **Modules** | `src.__main__.run_interactive_mode` | + +| Test | Behavior | +|------|----------| +| Ctrl+C | Farewell, no traceback | +| Integration | Interrupt via `builtins.input` | + +--- + +### `test_interactive_filters.py` + +| | | +|---|---| +| **Modules** | `src.memory.filters` | + +| Class | Behavior | +|-------|----------| +| `TestFollowUpParsing` | Year filters, focus, compare, export, new query | +| `TestPaperFiltering` | Year filter, keyword boost, report subset | + +--- + +### `test_phase3_extensibility.py` + +| | | +|---|---| +| **Modules** | registry, events, API scaffold, provider/fulltext stubs | + +| Class | Behavior | +|-------|----------| +| `TestPhase3ProviderStubs` | PubMed/Core/DBLP registered, disabled, NIE | +| `TestFullTextScaffold` | PDF downloader & RAG index stubs | +| `TestPluginRegistry` | Bootstrap registers providers + stages | +| `TestPipelineEvents` | `StageEventCollector` fires start/complete | +| `TestPdfReadyHtml` | Print CSS, A4 `@page` | +| `TestApiScaffold` | FastAPI optional import | + +--- + +## Cross-cutting patterns + +| Pattern | Where used | +|---------|------------| +| Stub pipeline stages | `EchoStage`, `RetrievalStub`, `NamedEchoStage` | +| Provider stubs | `SuccessProvider` / `FailingProvider` | +| `mock_pipeline_result` | `tests/helpers/pipeline_mocks.py` | +| `FixedEmbeddingProvider` | Deterministic vectors for ranking/relevance/quality | +| `monkeypatch` HDBSCAN | Force all-noise labels | +| `capsys` | CLI output assertions | +| Parametrized multi-domain | `test_research_quality.py` | +| `@pytest.mark.slow` subprocess | `test_complete_workflow.py` | +| `@pytest.mark.asyncio` | Async stage/pipeline/memory tests | + +--- + +## Coverage gaps (for docs) + +| Gap | Detail | +|-----|--------| +| Query understanding | No dedicated unit test file | +| API routes | Only scaffold test in `test_phase3_extensibility.py` | +| `EnhancedResponseHandler` | Subcomponents tested, not end-to-end | +| Live LLM integration | All LLM tests mock Pydantic AI | +| Subprocess tests | `@pytest.mark.slow`; may skip in CI | + +## Running tests + +```bash +pipenv install --dev +pipenv run pytest +pipenv run pytest tests/test_research_quality.py -v +pipenv run pytest -m "not slow" +``` diff --git a/docs/api/endpoints.md b/docs/api/endpoints.md new file mode 100644 index 0000000..55baaec --- /dev/null +++ b/docs/api/endpoints.md @@ -0,0 +1,225 @@ +# API Endpoints + +Route definitions in `src/api/app.py`. All routes are async. Request/response models use Pydantic v2. + +Base URL (local default): `http://127.0.0.1:8000` + +Interactive OpenAPI UI: `http://127.0.0.1:8000/docs` + +## GET `/health` + +Returns service status and registered plugin names. + +### Response schema + +```json +{ + "status": "ok", + "providers": ["arxiv", "core", "crossref", "dblp", "openalex", "pubmed", "semantic_scholar"], + "stages": [ + "citation_export", + "clustering", + "deduplication", + "gap_analysis", + "query_expansion", + "query_understanding", + "ranking", + "relevance_scoring", + "report_generation", + "retrieval", + "synthesis" + ] +} +``` + +| Field | Type | Description | +|-------|------|-------------| +| `status` | string | Always `"ok"` when the app is running | +| `providers` | string[] | Registered retrieval provider keys (includes stubs) | +| `stages` | string[] | Registered pipeline stage names | + +### Example + +```bash +curl -s http://127.0.0.1:8000/health | jq +``` + +Use for load-balancer probes and verifying `bootstrap_default_plugins()` ran successfully. + +--- + +## GET `/providers` + +Lists registered retrieval provider names only. + +### Response schema + +```json +{ + "providers": ["arxiv", "core", "crossref", "dblp", "openalex", "pubmed", "semantic_scholar"] +} +``` + +### Example + +```bash +curl -s http://127.0.0.1:8000/providers | jq '.providers' +``` + +!!! warning "Stub providers listed" + PubMed, CORE, and DBLP appear in the list but raise `NotImplementedError` on search when enabled. See [Provider matrix](../retrieval/provider-matrix.md). + +--- + +## POST `/research` + +Runs the full research pipeline for a query and returns structured report data plus rendered output. + +### Request schema + +```json +{ + "query": "transformer attention mechanisms", + "format": "json", + "export": ["bibtex", "apa"] +} +``` + +| Field | Type | Required | Constraints | +|-------|------|----------|-------------| +| `query` | string | **Yes** | Min length 1 | +| `format` | string | No | Default `"json"`. Pattern: `markdown` \| `json` \| `html` | +| `export` | string[] \| null | No | Citation export formats passed to `render_report_output` | + +!!! info "No PDF format" + Unlike the CLI, the API does not accept `format: pdf`. Use `html` and print-to-PDF client-side, or call the CLI with `--format pdf`. + +### Success response schema + +```json +{ + "query": "transformer attention mechanisms", + "format": "markdown", + "partial": false, + "warnings": [], + "duration_ms": 7420.5, + "session_id": "abc123", + "metrics": { + "stages": {}, + "retrieval": {}, + "llm_tokens_in": 0, + "llm_tokens_out": 0 + }, + "report": { + "query": "transformer attention mechanisms", + "executive_summary": "...", + "papers": [], + "themes": [], + "gaps": [], + "metadata": {} + }, + "rendered": "# Research Report\n\n..." +} +``` + +| Field | Type | Description | +|-------|------|-------------| +| `partial` | boolean | `true` if any stage returned partial success | +| `warnings` | string[] | Human-readable stage warnings | +| `duration_ms` | number | Wall-clock pipeline duration | +| `session_id` | string \| null | Session identifier from pipeline context | +| `metrics` | object | Aggregated pipeline metrics | +| `report` | object | Full `EnhancedResearchReport` JSON | +| `rendered` | string or object | Markdown/HTML string, or JSON dict when `format=json` | + +### Example: JSON report + +```bash +curl -s -X POST http://127.0.0.1:8000/research \ + -H "Content-Type: application/json" \ + -d '{"query": "graph neural networks survey", "format": "json"}' \ + | jq '.report.executive_summary' +``` + +### Example: Markdown rendered output + +```bash +curl -s -X POST http://127.0.0.1:8000/research \ + -H "Content-Type: application/json" \ + -d '{"query": "reinforcement learning robotics", "format": "markdown"}' \ + | jq -r '.rendered' +``` + +### Example: With citation export + +```bash +curl -s -X POST http://127.0.0.1:8000/research \ + -H "Content-Type: application/json" \ + -d '{ + "query": "diffusion models", + "format": "json", + "export": ["bibtex", "apa"] + }' | jq '.rendered' +``` + +The `export` list is forwarded to `render_report_output()` alongside the main format. Exact export keys in the response depend on the reporting layer — see [Output formats](../user-guide/output-formats.md). + +### Error responses + +| Status | Cause | Body | +|--------|-------|------| +| 422 | Invalid request (empty query, bad format) | FastAPI validation detail | +| 500 | Unhandled pipeline exception | `{"detail": ""}` | + +Pipeline stages that fail partially still return **200** with `partial: true` and populated `warnings` — only uncaught exceptions produce 500. + +### Internal flow + +```mermaid +sequenceDiagram + participant Client + participant API as FastAPI /research + participant Pipe as ResearchPipeline + participant Render as render_report_output + + Client->>API: POST {query, format, export} + API->>Pipe: build_pipeline(settings).execute(query) + Pipe-->>API: PipelineResult + API->>API: Extract EnhancedResearchReport + API->>Render: report, format, partial, warnings, export + Render-->>API: rendered output + API-->>Client: JSON envelope +``` + +Report extraction fallback: if `result.output` is not an `EnhancedResearchReport`, the handler checks `result.artifacts["enhanced_report"]`, then creates an empty report with the query string. + +## Programmatic usage + +```python +import asyncio +from httpx import ASGITransport, AsyncClient + +from src.api.app import create_app +from src.config.settings import AppSettings + +async def main() -> None: + app = create_app(AppSettings()) + transport = ASGITransport(app=app) + async with AsyncClient(transport=transport, base_url="http://test") as client: + response = await client.post( + "/research", + json={"query": "test query", "format": "json"}, + ) + data = response.json() + assert data["query"] == "test query" + +asyncio.run(main()) +``` + +Requires `httpx` (not a core dependency) for ASGI testing. + +## Related pages + +- [API overview](index.md) — install, CLI comparison, limitations +- [Output formats](../user-guide/output-formats.md) — format details and CLI parity +- [Architecture overview](../architecture/overview.md) — what the pipeline runs diff --git a/docs/api/index.md b/docs/api/index.md new file mode 100644 index 0000000..3ded0a7 --- /dev/null +++ b/docs/api/index.md @@ -0,0 +1,108 @@ +# API Overview + +The HTTP API is an **optional** FastAPI layer over the full 11-stage research pipeline. It is not installed with the core application. + +Source: `src/api/app.py`, `src/api/__init__.py`. + +## Install and run + +```bash +pipenv install fastapi uvicorn +pipenv run uvicorn src.api.app:create_app --factory --reload +``` + +The app listens on `http://127.0.0.1:8000` by default. OpenAPI docs: `http://127.0.0.1:8000/docs`. + +!!! info "Not in core Pipfile" + `fastapi` and `uvicorn` are intentionally excluded from `[packages]` in `Pipfile`. Install them when you need the API. + +## CLI vs API pipeline + +| Aspect | CLI batch (`python -m src "query"`) | API (`POST /research`) | +|--------|-------------------------------------|-------------------------| +| Pipeline builder | `run_research_helper` shortcut | `build_pipeline()` full pipeline | +| Default providers | OpenAlex + Semantic Scholar only | All providers enabled in config | +| Output | stdout / `-o` file | JSON response with `report` + `rendered` | +| Formats | markdown, json, html, pdf | markdown, json, html only | +| Session memory | `--session` flag | Not exposed | + +!!! info "Full pipeline on API" + The API uses `build_pipeline(resolved_settings)` — the same composable pipeline as programmatic use. Enable additional retrieval providers via config. See [CLI vs API](../user-guide/cli-vs-api.md) and [Retrieval overview](../retrieval/overview.md). + +## Application factory + +`create_app(settings=None)` in `src/api/app.py`: + +1. Imports FastAPI (raises `ImportError` with install hint if missing) +2. Calls `bootstrap_default_plugins()` — registers providers and stages +3. Resolves settings via argument or `get_settings()` +4. Mounts `/health`, `/providers`, `/research` routes + +Pass custom settings for tests or isolated deployments: + +```python +from src.api.app import create_app +from src.config.settings import AppSettings + +app = create_app(AppSettings.from_yaml()) +``` + +## Configuration + +The API reads the same settings as the CLI: + +- YAML in `config/` +- `.env` file +- `RA_*` environment variables + +No API-specific config keys exist. Set LLM provider, retrieval toggles, and pipeline flags before starting uvicorn: + +```bash +export RA_LLM__PROVIDER=openai +export OPENAI_API_KEY=sk-... +pipenv run uvicorn src.api.app:create_app --factory +``` + +See [Environment variables](../configuration/environment-variables.md). + +## Response shape (summary) + +`POST /research` returns: + +| Field | Description | +|-------|-------------| +| `query` | Echo of request query | +| `format` | Requested output format | +| `partial` | `true` if any stage failed partially | +| `warnings` | Stage warning messages | +| `duration_ms` | Total pipeline time | +| `session_id` | Pipeline session identifier | +| `metrics` | Stage/retrieval/LLM metrics dict | +| `report` | `EnhancedResearchReport` as JSON | +| `rendered` | Format-specific string (markdown/html) or dict (json) | + +Full schemas and examples: [Endpoints](endpoints.md). + +## Limitations + +| Limitation | Detail | +|------------|--------| +| No `pdf` format | Request `format` pattern allows `markdown\|json\|html` only | +| No streaming | Synchronous await of full pipeline per request | +| No auth | No built-in authentication or rate limiting | +| 500 on pipeline error | Unhandled exceptions become `HTTPException(500)` | +| `export` param | Accepted but behavior depends on `render_report_output` — see endpoints doc | + +## Testing + +API scaffold tests live in `tests/test_phase3_extensibility.py`. Install FastAPI first (see [Install and run](#install-and-run)), then run: + +`pipenv run pytest tests/test_phase3_extensibility.py::TestApiScaffold -v` + +Full pytest reference: [Testing](../development/testing.md). + +## Related pages + +- [Endpoints](endpoints.md) — route reference with JSON examples +- [Architecture overview](../architecture/overview.md) — pipeline stages +- [Local development setup](../development/local-setup.md) — dev environment diff --git a/docs/architecture/artifacts.md b/docs/architecture/artifacts.md new file mode 100644 index 0000000..6f1c8d2 --- /dev/null +++ b/docs/architecture/artifacts.md @@ -0,0 +1,90 @@ +# Artifact Registry + +Pipeline stages communicate through two mechanisms: + +1. **Sequential `data` chain** — each stage's output becomes the next stage's input. +2. **Shared artifact store** — `PipelineContext.set_artifact()` / `get_artifact()` for cross-stage data that does not fit the linear chain. + +Source: `src/core/context.py`, `src/core/pipeline.py`, stage modules under `src/research/`, `src/retrieval/`, `src/analysis/`, `src/reporting/`. + +## Master artifact map + +| Artifact key | Type | Set by | Read by | +|--------------|------|--------|---------| +| `query_understanding` | `QueryUnderstandingResult` | query_understanding | relevance_scoring | +| `expanded_queries` | `ExpandedQuerySet` | query_expansion | — | +| `cached_papers` | `list[dict]` | orchestrator (`initial_artifacts`) | retrieval | +| `retrieved_papers` | `list[RetrievedPaper]` | retrieval | synthesis (recovery) | +| `deduplication_stats` | `dict[str, int]` | deduplication | — | +| `query_embedding` | `list[float]` | ranking | relevance_scoring, clustering | +| `paper_embeddings` | `dict[str, list[float]]` | ranking | relevance_scoring, clustering | +| `ranked_papers` | `list[RankedPaper]` | ranking, relevance_scoring, synthesis | synthesis, citation_export, report_generation | +| `relevance_filter_reasons` | `dict[str, str]` | relevance_scoring | — | +| `paper_clusters` | `list[PaperCluster]` | clustering | synthesis, gap_analysis, report_generation | +| `paper_extractions` | `list[PaperExtraction]` | synthesis | synthesis (recovery) | +| `paper_analyses` | `list[PaperAnalysis]` | synthesis | report_generation | +| `synthesis_result` | `SynthesisResult` | synthesis | gap_analysis, report_generation | +| `gap_analysis` | `GapAnalysisResult` | gap_analysis | report_generation | +| `citation_exports` | `dict[str, str]` | citation_export | report_generation (via `data` param) | +| `citation_index` | `dict[str, str]` | citation_export | report_generation | +| `enhanced_report` | `EnhancedResearchReport` | report_generation | pipeline result, API, CLI | + +## Exported artifacts + +`ResearchPipeline.execute()` exports a subset of artifacts in `ResearchPipelineResult.artifacts`: + +``` +ranked_papers, retrieved_papers, paper_analyses, paper_clusters, +synthesis_result, gap_analysis, citation_exports, citation_index, enhanced_report +``` + +Internal-only artifacts (`query_understanding`, embeddings, `deduplication_stats`, etc.) are available in debug dumps when `debug_enabled` is set. + +## Pre-populated artifacts + +### Session cache bypass + +When memory caching is enabled and a cache hit occurs, `run_research_with_result()` in `src/retrieval/orchestrator.py` pre-populates: + +```python +initial_artifacts = {"cached_papers": cache_hit_papers} +``` + +The retrieval stage detects `cached_papers`, skips provider calls, and sets `retrieved_papers` directly. + +## Embedding artifacts + +Ranking stage stores embeddings via `store_ranking_embedding_result()` in `src/research/embedding_context.py`: + +- `query_embedding` — embedding vector for the user query +- `paper_embeddings` — map of `paper_id` → embedding vector + +These are consumed by relevance_scoring (adaptive embedding floor) and clustering (HDBSCAN input), avoiding redundant embedding computation. + +## Artifact vs data chain + +Some values exist in **both** places by design: + +| Value | On data chain | In artifacts | +|-------|---------------|--------------| +| Ranked papers | Passed through relevance_scoring → clustering | Also stored/updated as `ranked_papers` | +| Synthesis | Passed to gap_analysis | Also stored as `synthesis_result` | +| Citation exports | Passed to report_generation | Also stored as `citation_exports` | + +Downstream stages that need data from earlier stages (e.g., report_generation reading `paper_clusters` while receiving citation exports on the chain) rely on the artifact store. + +## Debug visibility + +When `RA_PIPELINE__DEBUG=true` (or `debug_enabled` in config), the pipeline writes `logs/debug/pipeline_{session_id}_{timestamp}.json` containing: + +- Stage result summaries (duration, warnings, partial flags) +- Artifact **key names** (not full values — values are listed by key only in `to_debug_dict()`) +- Pipeline metrics + +See [Logging and debug](../operations/logging-and-debug.md). + +## Related pages + +- [Pipeline stages](pipeline-stages.md) — stage index with artifact columns +- [Data model](data-model.md) — Pydantic type definitions +- Per-stage artifact details in [Stage deep dives](stages/query-understanding.md) diff --git a/docs/architecture/data-model.md b/docs/architecture/data-model.md new file mode 100644 index 0000000..a435123 --- /dev/null +++ b/docs/architecture/data-model.md @@ -0,0 +1,171 @@ +# Data Model + +All pipeline types are Pydantic `BaseModel` classes defined in `src/retrieval/models.py`. Stages pass typed objects through the sequential `data` chain and store additional objects in the artifact store. + +## Type flow through the pipeline + +| Stage | `data` input | `data` output | +|-------|--------------|---------------| +| query_understanding | `str` | `QueryUnderstandingResult` | +| query_expansion | `str \| QueryUnderstandingResult` | `ExpandedQuerySet` | +| retrieval | `ExpandedQuerySet` | `list[RetrievedPaper]` | +| deduplication | `list[RetrievedPaper]` | `list[RetrievedPaper]` | +| ranking | `list[RetrievedPaper]` | `list[RankedPaper]` | +| relevance_scoring | `list[RankedPaper]` | `list[RankedPaper]` | +| clustering | `list[RankedPaper]` | `list[PaperCluster]` | +| synthesis | `list[PaperCluster]` | `SynthesisResult` | +| gap_analysis | `SynthesisResult` | `GapAnalysisResult` | +| citation_export | `GapAnalysisResult` | `dict[str, str]` | +| report_generation | `dict[str, str]` | `EnhancedResearchReport` | + +## Core types + +### RetrievedPaper + +Paper retrieved from a scholarly API provider. + +| Field | Type | Notes | +|-------|------|-------| +| `title` | `str` | Required | +| `abstract` | `Optional[str]` | May be absent for metadata-only records | +| `year` | `Optional[int]` | Publication year | +| `venue` | `Optional[str]` | Journal or conference | +| `url` | `Optional[str]` | Landing page | +| `doi` | `Optional[str]` | Preferred stable ID | +| `provider` | `str` | Source provider name (alias: `source`) | +| `citation_count` | `Optional[int]` | When available from provider | +| `authors` | `list[str]` | Author names | +| `keywords` | `list[str]` | Subject keywords | +| `embedding_id` | `Optional[str]` | Internal embedding cache key | +| `raw_metadata` | `dict[str, Any]` | Provider-specific payload | + +**Computed properties:** + +- `source` — backward-compatible alias for `provider` +- `paper_id` — stable ID: `doi` → `url` → `title` + +### RankedPaper + +Wraps a `RetrievedPaper` with ranking signals. + +| Field | Type | Notes | +|-------|------|-------| +| `paper` | `RetrievedPaper` | Underlying paper | +| `rank_score` | `float` | Composite ranking score | +| `score_breakdown` | `dict[str, float]` | Per-signal contributions (embedding, citations, recency, etc.) | + +### QueryUnderstandingResult + +Structured query analysis from stage 1. + +| Field | Type | Notes | +|-------|------|-------| +| `intent` | `str` | `literature_review`, `comparison`, or `gap_analysis` | +| `constraints` | `dict[str, Any]` | Year filters (`years`, `min_year`, `max_year`) | +| `key_concepts` | `list[str]` | Extracted core concepts | + +### ExpandedQuerySet + +Query variants for multi-query retrieval. + +| Field | Type | Notes | +|-------|------|-------| +| `original` | `str` | User query | +| `variants` | `list[str]` | Rephrased search strings | +| `sub_questions` | `list[str]` | Decomposed sub-queries | + +### PaperCluster + +Thematic grouping from clustering stage. + +| Field | Type | Notes | +|-------|------|-------| +| `theme` | `str` | Cluster label | +| `summary` | `str` | Brief theme description | +| `paper_ids` | `list[str]` | Member `paper_id` values | + +### PaperExtraction + +Per-paper structured extraction (synthesis Pass A). Stored as artifact `paper_extractions`. + +| Field | Type | +|-------|------| +| `paper_id`, `title` | identifiers | +| `methodology`, `datasets`, `benchmarks`, `limitations`, `findings` | `list[str]` | + +### PaperAnalysis + +Per-paper analysis in the final report. + +| Field | Type | +|-------|------| +| `paper_id`, `title`, `year`, `venue`, `url`, `doi` | metadata | +| `key_points`, `why_relevant` | `list[str]` | + +### SynthesisResult + +Cross-paper synthesis (Pass B). + +| Field | Type | +|-------|------| +| `agreements`, `disagreements`, `trends`, `gaps` | `list[str]` | +| `datasets`, `methodologies` | `list[str]` | + +### GapAnalysisResult + +Prioritized research gaps and opportunities. + +| Field | Type | +|-------|------| +| `gaps`, `opportunities`, `underexplored_areas` | `list[str]` | + +### EnhancedResearchReport + +Final pipeline output — the primary deliverable. + +| Field | Type | Notes | +|-------|------|-------| +| `query` | `str` | Original user query | +| `executive_summary` | `str` | High-level overview | +| `papers` | `list[PaperAnalysis]` | Per-paper analyses | +| `clusters` | `list[PaperCluster]` | Thematic groupings | +| `synthesis` | `Optional[SynthesisResult]` | Cross-paper synthesis | +| `gap_analysis` | `Optional[GapAnalysisResult]` | Gap/opportunity analysis | +| `gaps` | `list[str]` | Flat gap list (legacy convenience) | +| `timeline` | `list[str]` | Chronological trend lines | +| `citation_index` | `dict[str, str]` | Citation key → paper ID | +| `exports` | `dict[str, str]` | BibTeX, CSL, etc. | + +`to_research_report()` converts to the legacy `ResearchReport` shape used by some renderers. + +## Pipeline result wrapper + +`ResearchPipelineResult` (`src/core/pipeline.py`) wraps the full run: + +| Field | Purpose | +|-------|---------| +| `query` | Original query | +| `output` | Final stage output (`EnhancedResearchReport`) | +| `session_id` | Research session UUID | +| `stage_results` | Per-stage duration, warnings, metrics | +| `warnings` | Accumulated pipeline warnings | +| `partial` | `true` if any stage returned partial/heuristic output | +| `duration_ms` | Total wall time | +| `metrics` | Aggregated pipeline metrics | +| `artifacts` | Subset of artifact store exported to callers | + +Exported artifact keys: `ranked_papers`, `retrieved_papers`, `paper_analyses`, `paper_clusters`, `synthesis_result`, `gap_analysis`, `citation_exports`, `citation_index`, `enhanced_report`. + +## Session and context types + +**`ResearchSession`** (`src/core/context.py`) — in-memory session with UUID, timestamps, and optional `context_json` for interactive follow-ups. + +**`PipelineContext`** — shared state per run: query, resolved config, session, metrics, warnings, stage results, and the artifact dictionary. + +**`StageResult[T]`** — generic stage output wrapper with `output`, `duration_ms`, `metrics`, `warnings`, and `partial` flag. + +## Related pages + +- [Artifacts](artifacts.md) — which types are stored as artifacts vs passed on the data chain +- [Pipeline stages](pipeline-stages.md) — stage index +- [Output formats](../user-guide/output-formats.md) — how `EnhancedResearchReport` is rendered diff --git a/docs/architecture/llm-layer.md b/docs/architecture/llm-layer.md new file mode 100644 index 0000000..95fb11e --- /dev/null +++ b/docs/architecture/llm-layer.md @@ -0,0 +1,155 @@ +# LLM Layer + +The LLM layer provides chat-model agents for query expansion, synthesis, and gap analysis. It is separate from the **embedding model** used by ranking, deduplication, relevance scoring, and clustering. + +Source: `src/models/`, `src/config/resolve_llm_features.py`, `src/config/model_selection.py`. + +## Resolution phases + +LLM behavior is determined in three phases before any stage makes a call: + +```mermaid +flowchart TD + LOAD[Settings load] --> RESOLVE[resolve_effective_settings] + RESOLVE --> FACTORY[AgentFactory at call site] + FACTORY --> MODEL[Provider.create_model] + MODEL --> AGENT[Agent with role prompt] +``` + +### Phase 1: Settings load + +`AppSettings` merges kwargs → `RA_*` env vars → `.env` → YAML → defaults. Key LLM fields: + +| Field | Default | Env override | +|-------|---------|--------------| +| `llm.provider` | `"ollama"` | `RA_LLM__PROVIDER` | +| `llm.model` | `"auto"` | `RA_LLM__MODEL` | +| `llm.base_url` | `"http://localhost:11434"` | `RA_LLM__BASE_URL` | +| `llm.api_key` | `None` | `RA_LLM__API_KEY` | +| `synthesis.llm_mode` | `"auto"` | `RA_SYNTHESIS__LLM_MODE` | +| `query_expansion.llm_mode` | `"auto"` | `RA_QUERY_EXPANSION__LLM_MODE` | + +### Phase 2: Feature flag resolution + +`resolve_effective_settings()` runs at pipeline start (`ResearchPipeline.execute()`). It sets `synthesis.llm_enabled` and `query_expansion.llm_enabled` on a copy of settings passed to all stages via `ctx.config`. + +**Precedence per feature** (synthesis, query_expansion): + +1. `RA_{SECTION}__LLM_ENABLED` env override (`true`/`false`/`1`/`0`) +2. `llm_mode: on` → enabled; `llm_mode: off` → disabled +3. `llm_mode: auto` → rules below + +**Auto-mode rules:** + +| Provider | synthesis LLM | query_expansion LLM | +|----------|---------------|---------------------| +| `openai`, `anthropic` | Always enabled | Always enabled | +| `ollama` | Enabled if `ollama_models.yaml` entry has `synthesis.llm_enabled: true` for resolved model | Same catalog hint | +| Other | Disabled | Disabled | + +!!! warning "Default quality" + With Ollama `llama3.2:3b` (fallback) and `llm_mode: auto`, both synthesis and query expansion LLM are **off**. Reports use heuristics. See [Heuristic vs LLM](../llm/heuristic-vs-llm.md). + +When provider is Ollama and model resolves from catalog, `max_llm_papers` may be updated from catalog hints (e.g., 8B model → 5 papers). + +### Phase 3: Model name resolution + +When `llm.model` is `"auto"` or empty, `resolve_llm_model_name()` in `src/config/model_selection.py`: + +1. Loads `config/ollama_models.yaml` +2. If `auto_select: true`, detects system resources (RAM, disk, swap) +3. Picks highest-priority model that fits resources +4. Falls back to catalog `fallback` model if none fit + +Explicit model names (CLI, env, or YAML) skip auto-selection. + +### Phase 4: Provider and agent creation + +`AgentFactory` (`src/models/factory.py`) resolves config, instantiates the provider, creates a pydantic-ai `Agent` with the role system prompt. + +## Provider registry + +| Provider key | Class | Backend | +|--------------|-------|---------| +| `ollama` | `OllamaProvider` | OpenAI-compatible API at normalized `base_url` | +| `openai` | `OpenAIProviderImpl` | pydantic-ai OpenAI model | +| `anthropic` | `AnthropicProviderImpl` | pydantic-ai Anthropic model | + +Register custom providers with `register_llm_provider()`. + +## API key resolution + +| Provider | Key sources (priority) | +|----------|------------------------| +| Ollama | `RA_LLM__API_KEY` → `OLLAMA_API_KEY` → default `"ollama"` | +| OpenAI | `RA_LLM__API_KEY` → `OPENAI_API_KEY` (required) | +| Anthropic | `RA_LLM__API_KEY` → `ANTHROPIC_API_KEY` (required) | + +## Base URL normalization + +`normalize_openai_base_url()` appends `/v1` if missing. The code default is `http://localhost:11434`; `.env.example` may show `http://localhost:11434/v1` — both resolve to the same endpoint. + +## Agent roles + +| Role | Enum | Stage | Purpose | +|------|------|-------|---------| +| EXPANSION | `AgentRole.EXPANSION` | query_expansion | JSON variants + sub_questions | +| EXTRACTION | `AgentRole.EXTRACTION` | synthesis (Pass A) | Per-paper structured extraction | +| SYNTHESIS | `AgentRole.SYNTHESIS` | synthesis (Pass B) | Cross-paper synthesis JSON | +| GAP_ANALYSIS | `AgentRole.GAP_ANALYSIS` | gap_analysis | Gaps/opportunities JSON | +| ANALYSIS | `AgentRole.ANALYSIS` | `src/analysis/llm.py` only | Legacy module-level agent | + +## Runtime call sites + +```mermaid +flowchart TD + RESOLVED[ctx.config after resolve_effective_settings] + + RESOLVED --> QE{query_expansion.llm_enabled?} + QE -->|yes| QE_AGENT[AgentFactory EXPANSION] + QE -->|no| QE_SKIP[heuristics only] + + RESOLVED --> SY{synthesis.llm_enabled?} + SY -->|yes| SY_A[EXTRACTION up to max_llm_papers] + SY_A --> SY_B[SYNTHESIS collective] + SY -->|no| SY_H[heuristic extraction + synthesis] + + RESOLVED --> GA{synthesis.llm_enabled?} + GA -->|yes| GA_AGENT[GAP_ANALYSIS structured] + GA -->|no| GA_H[heuristic from synthesis fields] +``` + +!!! info "Gap analysis coupling" + Gap analysis LLM is gated by **`synthesis.llm_enabled`**, not a separate `gap_analysis.llm_mode`. + +## Embedding model (separate) + +The chat LLM and embedding model are independent: + +| Component | Config | Used by | +|-----------|--------|---------| +| Chat LLM | `llm.*` | query_expansion, synthesis, gap_analysis | +| Embeddings | `embedding.*` | deduplication, ranking, relevance_scoring, clustering | + +Embedding provider: sentence-transformers via `src/embeddings/`. If not installed, embedding-dependent features degrade gracefully with warnings. + +## Non-pipeline LLM module + +`src/analysis/llm.py` creates a module-level `analysis_agent` at import via `AgentFactory()` — **not** part of the 11-stage pipeline. Used by legacy/orchestrator helper paths. + +## Known gaps + +| Issue | Detail | +|-------|--------| +| `llm.timeout_seconds` | Configured but unused; stage timeouts are pipeline-level | +| `llm.temperature` | Not passed to pydantic-ai model constructors | +| `expand_query_llm` | Uses `get_settings().llm` instead of `ctx.config.llm` | +| `analysis/llm.py` | Global agent at import — separate from pipeline | + +## Related pages + +- [Ollama](../llm/ollama.md) — auto-select, catalog, setup integration +- [Cloud providers](../llm/cloud-providers.md) — OpenAI, Anthropic +- [Heuristic vs LLM](../llm/heuristic-vs-llm.md) — quality tradeoffs +- [Synthesis stage](stages/synthesis.md) — two-pass LLM flow +- [Environment variables](../configuration/environment-variables.md) diff --git a/docs/architecture/overview.md b/docs/architecture/overview.md new file mode 100644 index 0000000..242c7df --- /dev/null +++ b/docs/architecture/overview.md @@ -0,0 +1,100 @@ +# Architecture Overview + +The AI Research Assistant is a local-first, multi-stage Python research pipeline. A user query flows through eleven sequential stages that retrieve scholarly papers, rank and cluster them, synthesize findings, and assemble a structured report. + +## Entry points + +| Entry point | Module | Pipeline used | +|-------------|--------|---------------| +| CLI (`python -m src "query"`) | `src/__main__.py` → `run_research_helper()` | Full 11-stage pipeline, but **hardcodes OpenAlex + Semantic Scholar only** | +| CLI (programmatic) | `run_research()` / `run_research_with_result()` | Full pipeline with loaded `AppSettings` | +| FastAPI | `src/api/app.py` → `POST /research` | Full pipeline with request/config overrides | + +!!! info "CLI vs full pipeline" + `run_research_helper()` builds a minimal `AppSettings` with only OpenAlex and Semantic Scholar enabled. To use arXiv, CrossRef, or other providers, call `run_research()` with a custom config or use the API. See [Retrieval overview](../retrieval/overview.md). + +## End-to-end flow + +```mermaid +flowchart TD + CLI["CLI python -m src"] --> Orch["orchestrator.py"] + API["FastAPI POST /research"] --> Orch + Orch --> Build["build_pipeline()"] + Build --> Pipe["ResearchPipeline.execute()"] + Config["config/*.yaml + RA_* env"] --> Pipe + Pipe --> Resolve["resolve_effective_settings()"] + Resolve --> Stages["11 sequential stages"] + Stages --> Report["EnhancedResearchReport"] + Report --> Out["markdown / json / html / pdf-ready"] +``` + +## Project structure + +``` +src/ +├── __main__.py # CLI entry +├── api/app.py # Optional FastAPI layer +├── config/ # AppSettings, YAML loading, LLM resolution +├── core/ # Pipeline, context, registry, stage recovery +├── research/ # Query understanding, expansion, ranking, clustering +├── retrieval/ # Providers, retrieval stage, deduplication +├── analysis/ # Synthesis, gap analysis +├── reporting/ # Citations, report assembly, markdown render +├── models/ # LLM provider factory (Ollama, OpenAI, Anthropic) +├── embeddings/ # Sentence-transformer embedding provider +└── memory/ # Session cache and persistence +``` + +Configuration lives in `config/*.yaml` and is overridden by `.env` and `RA_*` environment variables. See [Configuration precedence](../configuration/precedence.md). + +## Pipeline orchestration + +`build_pipeline()` in `src/retrieval/orchestrator.py` constructs a `ResearchPipeline` with eleven stage instances in fixed order. The pipeline is registered in `src/core/registry.py` for extensibility. + +**Execution model** (`src/core/pipeline.py`): + +1. `resolve_effective_settings()` runs once at pipeline start — resolves LLM feature flags and Ollama model hints. +2. Each enabled stage runs sequentially; stage output becomes the next stage's `data` input. +3. Stages also read/write a shared **artifact store** on `PipelineContext` for cross-stage data (embeddings, ranked papers, synthesis, etc.). +4. Disabled stages (`pipeline.enabled_stages.*`) are skipped entirely. +5. Timeouts default to 300 s per stage; synthesis uses 600 s (`pipeline.synthesis_timeout_seconds`). +6. On failure or timeout, `continue_on_stage_failure` (default `true`) triggers heuristic recovery via `src/core/stage_recovery.py`. +7. When `debug_enabled`, a JSON dump is written to `logs/debug/pipeline_*.json`. + +## Data flow summary + +```mermaid +flowchart LR + Q[query: str] --> QU[query_understanding] + QU -->|QueryUnderstandingResult| QE[query_expansion] + QE -->|ExpandedQuerySet| RT[retrieval] + RT -->|list RetrievedPaper| DD[deduplication] + DD -->|list RetrievedPaper| RK[ranking] + RK -->|list RankedPaper| RS[relevance_scoring] + RS -->|list RankedPaper| CL[clustering] + CL -->|list PaperCluster| SY[synthesis] + SY -->|SynthesisResult| GA[gap_analysis] + GA -->|GapAnalysisResult| CE[citation_export] + CE -->|dict exports| RG[report_generation] + RG -->|EnhancedResearchReport| OUT[output] +``` + +Side-channel artifacts (embeddings, analyses, citation index) are stored on `PipelineContext` and documented in [Artifacts](artifacts.md). + +## Key design decisions + +| Decision | Rationale | +|----------|-----------| +| Sequential stages with typed `data` chain | Simple debugging, clear stage boundaries, easy enable/disable | +| Shared artifact store | Embeddings and ranked papers needed by multiple downstream stages | +| Heuristic defaults for LLM stages | Fast local runs without GPU/API; quality tradeoff documented in [Heuristic vs LLM](../llm/heuristic-vs-llm.md) | +| Graceful degradation | Partial reports with warnings rather than hard failure on single-stage errors | +| Separate embedding model | Ranking, dedup, relevance, and clustering use `embedding.*` config — independent of chat LLM | + +## Related pages + +- [Pipeline stages](pipeline-stages.md) — stage index and config overview +- [Stage deep dives](stages/query-understanding.md) — per-stage reference +- [Artifacts](artifacts.md) — artifact key registry and producers/consumers +- [Data model](data-model.md) — Pydantic types through the pipeline +- [LLM layer](llm-layer.md) — provider factory, roles, and resolution diff --git a/docs/architecture/pipeline-stages.md b/docs/architecture/pipeline-stages.md new file mode 100644 index 0000000..169d23b --- /dev/null +++ b/docs/architecture/pipeline-stages.md @@ -0,0 +1,95 @@ +# Pipeline Stages + +The research pipeline runs eleven stages in fixed order. Each stage implements the `PipelineStage` protocol (`name` + `async run(ctx, data) → StageResult`). + +Built by `build_pipeline()` in `src/retrieval/orchestrator.py`. Registry keys in `src/core/registry.py`. + +## Stage order + +``` +query_understanding → query_expansion → retrieval → deduplication → ranking → +relevance_scoring → clustering → synthesis → gap_analysis → citation_export → report_generation +``` + +## Summary table + +| # | Stage | Class | LLM | Timeout | Deep dive | +|---|-------|-------|-----|---------|-----------| +| 1 | query_understanding | `QueryUnderstandingStage` | No | 300 s | [→](stages/query-understanding.md) | +| 2 | query_expansion | `QueryExpansionStage` | Optional | 300 s | [→](stages/query-expansion.md) | +| 3 | retrieval | `RetrievalStage` | No | 300 s | [→](stages/retrieval.md) | +| 4 | deduplication | `DeduplicationStage` | No | 300 s | [→](stages/deduplication.md) | +| 5 | ranking | `RankingStage` | No | 300 s | [→](stages/ranking.md) | +| 6 | relevance_scoring | `RelevanceScoringStage` | No | 300 s | [→](stages/relevance-scoring.md) | +| 7 | clustering | `ClusteringStage` | No | 300 s | [→](stages/clustering.md) | +| 8 | synthesis | `SynthesisStage` | Optional (two-pass) | **600 s** | [→](stages/synthesis.md) | +| 9 | gap_analysis | `GapAnalysisStage` | Optional | 300 s | [→](stages/gap-analysis.md) | +| 10 | citation_export | `CitationExportStage` | No | 300 s | [→](stages/citation-export.md) | +| 11 | report_generation | `ReportGenerationStage` | No | 300 s | [→](stages/report-generation.md) | + +Default timeouts from `pipeline.stage_timeout_seconds` (300) and `pipeline.synthesis_timeout_seconds` (600). + +## Enable/disable + +Each stage can be toggled via `pipeline.enabled_stages.{stage_name}` in YAML or environment. Disabled stages are skipped; downstream stages receive the last enabled stage's output unchanged. See [Stage toggles](../configuration/stage-toggles.md). + +## Execution behavior + +| Mechanism | Behavior | +|-----------|----------| +| Sequential `data` chain | Each stage output → next stage input | +| Artifact store | Cross-stage shared state on `PipelineContext` | +| LLM resolution | `resolve_effective_settings()` before first stage | +| Failure handling | `continue_on_stage_failure: true` (default) — heuristic recovery | +| Timeout recovery | `src/core/stage_recovery.py` — synthesis and gap_analysis have dedicated fallbacks | +| Progress events | `PipelineEventBus` emits stage start/complete for stderr progress reporter | + +## Stage recovery + +| Stage | Recovery on timeout/failure | +|-------|----------------------------| +| synthesis | Heuristic extraction + synthesis from ranked papers | +| gap_analysis | Heuristic gaps from synthesis fields + clusters | +| All others | Return prior `data` unchanged | + +## Data-flow diagram + +```mermaid +flowchart LR + Q[query: str] --> QU[query_understanding] + QU -->|QueryUnderstandingResult| QE[query_expansion] + QE -->|ExpandedQuerySet| RT[retrieval] + RT -->|list RetrievedPaper| DD[deduplication] + DD -->|list RetrievedPaper| RK[ranking] + RK -->|list RankedPaper| RS[relevance_scoring] + RS -->|list RankedPaper| CL[clustering] + CL -->|list PaperCluster| SY[synthesis] + SY -->|SynthesisResult| GA[gap_analysis] + GA -->|GapAnalysisResult| CE[citation_export] + CE -->|dict exports| RG[report_generation] + RG -->|EnhancedResearchReport| OUT[output] +``` + +## Config keys by stage + +| Stage | Primary config sections | +|-------|------------------------| +| query_understanding | — | +| query_expansion | `query_expansion.*` | +| retrieval | `retrieval.*`, `retrieval.providers.*` | +| deduplication | `deduplication.*`, `embedding.*` | +| ranking | `ranking.*`, `embedding.*` | +| relevance_scoring | `relevance_scoring.*` | +| clustering | `clustering.*`, `embedding.*`, `ranking.*` | +| synthesis | `synthesis.*`, `llm.*` | +| gap_analysis | `synthesis.llm_enabled` (gates LLM — no separate gap flag) | +| citation_export | — | +| report_generation | `relevance_scoring.*` (executive summary embedding floor) | + +## Related pages + +- [Architecture overview](overview.md) +- [Artifacts](artifacts.md) +- [Data model](data-model.md) +- [LLM layer](llm-layer.md) +- [Stage toggles](../configuration/stage-toggles.md) diff --git a/docs/architecture/stages/citation-export.md b/docs/architecture/stages/citation-export.md new file mode 100644 index 0000000..14b3064 --- /dev/null +++ b/docs/architecture/stages/citation-export.md @@ -0,0 +1,52 @@ +# Stage: citation_export + +Generates citation exports and a citation index from ranked papers. + +| | | +|---|---| +| **Class** | `CitationExportStage` | +| **Module** | `src/reporting/citations.py` | +| **Registry key** | `citation_export` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `GapAnalysisResult` | Passed through from gap_analysis | +| Input (artifact) | `ranked_papers` | Source papers for citation generation | +| Output (`data`) | `dict[str, str]` | Format name → export content | +| Artifacts written | `citation_exports`, `citation_index` | Used by report_generation | + +## Behavior + +1. Extracts `RetrievedPaper` objects from `ranked_papers` artifact +2. Builds citation index (citation key → paper ID) via `build_citation_index()` +3. Generates exports for all `SUPPORTED_FORMATS` (BibTeX, CSL JSON, etc.) via `generate_citation_exports()` + +Deterministic — no LLM calls. The gap analysis result on the data chain is not used directly; this stage reads papers from artifacts. + +## Configuration + +No stage-specific config keys. + +## LLM + +No — BibTeX/CSL via `generate_citation_exports`. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On failure, returns prior `data` unchanged. + +## Metrics + +- `citation_count` — number of entries in citation index + +## Related + +- [Previous: gap_analysis](gap-analysis.md) +- [Next: report_generation](report-generation.md) +- [Output formats](../../user-guide/output-formats.md) diff --git a/docs/architecture/stages/clustering.md b/docs/architecture/stages/clustering.md new file mode 100644 index 0000000..9780d9c --- /dev/null +++ b/docs/architecture/stages/clustering.md @@ -0,0 +1,57 @@ +# Stage: clustering + +Groups ranked papers into thematic clusters for synthesis and report structure. + +| | | +|---|---| +| **Class** | `ClusteringStage` | +| **Module** | `src/research/clustering.py` | +| **Registry key** | `clustering` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `list[RankedPaper]` | From relevance_scoring | +| Input (artifact) | `paper_embeddings` | Reused from ranking when available | +| Output (`data`) | `list[PaperCluster]` | Thematic groups | +| Artifacts written | `paper_clusters` | Read by synthesis, gap_analysis, report_generation | + +## Behavior + +Primary algorithm: **HDBSCAN** on paper embedding vectors when sentence-transformers is available and embeddings exist. + +Fallback: single keyword-based theme group when embeddings are unavailable or clustering produces no clusters. + +Each cluster includes a `theme` label, `summary`, and list of member `paper_ids`. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `clustering.min_cluster_size` | HDBSCAN minimum cluster size | +| `clustering.min_samples` | HDBSCAN density parameter | +| `embedding.*` | Embedding model (if re-computing) | +| `ranking.*` | Used by `ensure_ranked_papers` adapter | + +## LLM + +No — HDBSCAN + keyword fallback. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +Empty input returns empty cluster list. On failure, returns prior `data` unchanged. + +## Metrics + +- `num_clusters` + +## Related + +- [Previous: relevance_scoring](relevance-scoring.md) +- [Next: synthesis](synthesis.md) +- [Data model: PaperCluster](../data-model.md) diff --git a/docs/architecture/stages/deduplication.md b/docs/architecture/stages/deduplication.md new file mode 100644 index 0000000..e7c1f24 --- /dev/null +++ b/docs/architecture/stages/deduplication.md @@ -0,0 +1,57 @@ +# Stage: deduplication + +Removes duplicate papers from the retrieved set using metadata matching and optional embedding similarity. + +| | | +|---|---| +| **Class** | `DeduplicationStage` | +| **Module** | `src/retrieval/deduplication.py` | +| **Registry key** | `deduplication` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `list[RetrievedPaper]` | From retrieval | +| Output (`data`) | `list[RetrievedPaper]` | Deduplicated list | +| Artifacts written | `deduplication_stats` | Counts: input, metadata_removed, embedding_removed, output | + +## Behavior + +Two deduplication passes when enabled: + +1. **Metadata union-find** — groups papers by DOI, normalized title, or URL overlap; keeps highest-citation representative +2. **Embedding similarity** (optional) — when `enable_embedding_dedup` is true and sentence-transformers is installed, removes papers above `embedding_similarity_threshold` + +If deduplication is disabled (`deduplication.enabled: false`), papers pass through unchanged. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `deduplication.enabled` | Master toggle | +| `deduplication.enable_embedding_dedup` | Enable embedding-based dedup | +| `deduplication.embedding_similarity_threshold` | Cosine similarity cutoff | +| `embedding.*` | Embedding model config when embedding dedup runs | + +## LLM + +No — metadata union-find + optional embedding similarity. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On failure, returns prior `data` unchanged. + +## Metrics + +Full `deduplication_stats` dict passed as stage metrics. + +## Related + +- [Previous: retrieval](retrieval.md) +- [Next: ranking](ranking.md) +- [Data model: RetrievedPaper](../data-model.md) diff --git a/docs/architecture/stages/gap-analysis.md b/docs/architecture/stages/gap-analysis.md new file mode 100644 index 0000000..88d1a89 --- /dev/null +++ b/docs/architecture/stages/gap-analysis.md @@ -0,0 +1,65 @@ +# Stage: gap_analysis + +Expands synthesis gaps into prioritized research opportunities and underexplored areas. + +| | | +|---|---| +| **Class** | `GapAnalysisStage` | +| **Module** | `src/analysis/gap_analysis.py` | +| **Registry key** | `gap_analysis` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `SynthesisResult` | From synthesis (may be wrong type — recovered) | +| Input (artifacts) | `paper_clusters`, `synthesis_result` | Context for gap identification | +| Output (`data`) | `GapAnalysisResult` | Passed to citation_export | +| Artifacts written | `gap_analysis` | Read by report_generation | + +## Behavior + +### Heuristic mode (when `synthesis.llm_enabled` is false) + +Derives gaps from synthesis fields (`gaps`, `trends`, `disagreements`) and cluster themes. No LLM call. Adds warning: *"Skipped LLM gap analysis (synthesis.llm_enabled=false)"*. + +### LLM mode + +Calls `AgentRole.GAP_ANALYSIS` with structured JSON response containing gaps, opportunities, and underexplored areas. + +!!! info "LLM gate" + Gap analysis LLM is controlled by **`synthesis.llm_enabled`**, not a separate gap-analysis config flag. + +### Input recovery + +`resolve_synthesis_input()` recovers synthesis from artifacts if the data chain carries an unexpected type (e.g., cluster list from a skipped synthesis stage). + +## Configuration + +| Key | Purpose | +|-----|---------| +| `synthesis.llm_enabled` | Gates LLM gap analysis | +| `llm.*` | Provider/model when LLM enabled | + +## LLM + +Optional — `AgentRole.GAP_ANALYSIS` when `synthesis.llm_enabled`. Heuristic fallback otherwise. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +`recover_gap_analysis_output()` in `src/core/stage_recovery.py` on pipeline timeout. In-stage heuristic fallback on LLM exception. + +## Metrics + +- `gap_count` +- `opportunity_count` + +## Related + +- [Previous: synthesis](synthesis.md) +- [Next: citation_export](citation-export.md) +- [Data model: GapAnalysisResult](../data-model.md) diff --git a/docs/architecture/stages/query-expansion.md b/docs/architecture/stages/query-expansion.md new file mode 100644 index 0000000..d58380e --- /dev/null +++ b/docs/architecture/stages/query-expansion.md @@ -0,0 +1,63 @@ +# Stage: query_expansion + +Generates query variants and sub-questions to improve retrieval coverage. + +| | | +|---|---| +| **Class** | `QueryExpansionStage` | +| **Module** | `src/research/query_expansion.py` | +| **Registry key** | `query_expansion` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `str \| QueryUnderstandingResult` | Uses `key_concepts` when understanding result provided | +| Output (`data`) | `ExpandedQuerySet` | Passed to retrieval | +| Artifacts written | `expanded_queries` | Debug visibility only | + +## Behavior + +Two-phase expansion: + +1. **Heuristics always run** — synonym variants, concept permutations, sub-question templates +2. **Optional LLM pass** — when `query_expansion.llm_enabled`, calls `AgentRole.EXPANSION` for additional variants and sub-questions; merged with heuristic output (deduplicated, capped) + +When LLM is disabled or fails, heuristic-only expansion is returned. + +!!! warning "Settings inconsistency" + `expand_query_llm()` reads `get_settings().llm` rather than `ctx.config.llm`. If settings differ between pipeline config and global singleton, LLM expansion may use unexpected provider/model. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `query_expansion.llm_enabled` | Resolved at pipeline start from `llm_mode` + env | +| `query_expansion.llm_mode` | `auto` / `on` / `off` | +| `query_expansion.max_variants` | Cap on query variants | +| `query_expansion.max_sub_questions` | Cap on sub-questions | + +Env overrides: `RA_QUERY_EXPANSION__LLM_ENABLED`, `RA_QUERY_EXPANSION__LLM_MODE`. + +## LLM + +Optional — `AgentRole.EXPANSION` when enabled. Heuristics always produce baseline expansion. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On failure, returns prior `data` unchanged. + +## Metrics + +- `variant_count` +- `sub_question_count` + +## Related + +- [Previous: query_understanding](query-understanding.md) +- [Next: retrieval](retrieval.md) +- [LLM layer](../llm-layer.md) diff --git a/docs/architecture/stages/query-understanding.md b/docs/architecture/stages/query-understanding.md new file mode 100644 index 0000000..eb28566 --- /dev/null +++ b/docs/architecture/stages/query-understanding.md @@ -0,0 +1,48 @@ +# Stage: query_understanding + +Extracts structured intent, constraints, and key concepts from the raw user query. + +| | | +|---|---| +| **Class** | `QueryUnderstandingStage` | +| **Module** | `src/research/query_understanding.py` | +| **Registry key** | `query_understanding` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `str` | Raw query (initial pipeline input) | +| Output (`data`) | `QueryUnderstandingResult` | Passed to query_expansion | +| Artifacts written | `query_understanding` | Read by relevance_scoring | + +## Behavior + +Pure heuristic — no LLM calls. Uses regex and keyword extraction: + +- **Intent detection:** `literature_review` (default), `comparison` (compare/versus/vs), or `gap_analysis` (gap/opportunity keywords) +- **Year constraints:** explicit years, `after YYYY`, `before YYYY` +- **Key concepts:** extracted via `extract_core_concepts()` shared with query expansion + +## Configuration + +No stage-specific config keys. Always runs when enabled. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On failure, returns prior `data` unchanged (no dedicated recovery path). + +## Metrics + +- `intent` — detected intent string +- `concept_count` — number of key concepts + +## Related + +- [Next: query_expansion](query-expansion.md) +- [Data model: QueryUnderstandingResult](../data-model.md) +- [Pipeline stages](../pipeline-stages.md) diff --git a/docs/architecture/stages/ranking.md b/docs/architecture/stages/ranking.md new file mode 100644 index 0000000..40a8b29 --- /dev/null +++ b/docs/architecture/stages/ranking.md @@ -0,0 +1,65 @@ +# Stage: ranking + +Scores and ranks deduplicated papers using a weighted composite of multiple signals. + +| | | +|---|---| +| **Class** | `RankingStage` | +| **Module** | `src/research/ranking.py` | +| **Registry key** | `ranking` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `list[RetrievedPaper]` | From deduplication | +| Output (`data`) | `list[RankedPaper]` | Top-K by composite score | +| Artifacts written | `ranked_papers`, `query_embedding`, `paper_embeddings` | Embeddings reused downstream | + +## Behavior + +Computes a weighted composite score per paper: + +| Signal | Config weight key | +|--------|-------------------| +| Embedding similarity to query | `ranking.weights.embedding_similarity` | +| Citation count | `ranking.weights.citation_count` | +| Recency | `ranking.weights.recency` | +| Keyword overlap | `ranking.weights.keyword_overlap` | +| Venue quality | `ranking.weights.venue_quality` | + +Additional tuning: `domain_penalty_multiplier`, `outlier_embedding_gap`, `keyword_collision_max_sim`, `canonical_boost`. + +Results are sorted and truncated to `ranking.top_k`. Query and paper embeddings are stored via `store_ranking_embedding_result()` for reuse by relevance_scoring and clustering. + +If sentence-transformers is unavailable and embedding weight > 0, falls back to keyword-only ranking with a warning. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `ranking.top_k` | Maximum papers to pass downstream | +| `ranking.weights.*` | Signal weight overrides | +| `embedding.*` | Embedding model for similarity scoring | + +## LLM + +No. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On `ImportError`, retries with keyword-only fallback. On other failure, returns prior `data` unchanged. + +## Metrics + +- `top_score` — highest rank_score in output + +## Related + +- [Previous: deduplication](deduplication.md) +- [Next: relevance_scoring](relevance-scoring.md) +- [Artifacts: embedding storage](../artifacts.md) diff --git a/docs/architecture/stages/relevance-scoring.md b/docs/architecture/stages/relevance-scoring.md new file mode 100644 index 0000000..d751067 --- /dev/null +++ b/docs/architecture/stages/relevance-scoring.md @@ -0,0 +1,60 @@ +# Stage: relevance_scoring + +Filters ranked papers below composite relevance thresholds while ensuring a minimum corpus size. + +| | | +|---|---| +| **Class** | `RelevanceScoringStage` | +| **Module** | `src/research/relevance_scoring.py` | +| **Registry key** | `relevance_scoring` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `list[RankedPaper]` | From ranking | +| Input (artifacts) | `query_understanding`, `query_embedding`, `paper_embeddings` | Concept matching + embedding floor | +| Output (`data`) | `list[RankedPaper]` | Filtered list | +| Artifacts written | `ranked_papers` (updated), `relevance_filter_reasons` | Reasons for excluded papers | + +## Behavior + +For each ranked paper, evaluates: + +1. **Minimum rank score** — `relevance_scoring.min_rank_score` +2. **Embedding similarity floor** — `min_embedding_similarity`, optionally adaptive via `adaptive_embedding` +3. **Concept coverage** — key concepts from query understanding must appear in title/abstract + +Papers failing any check are removed. If the filtered set falls below `min_keep_papers`, the relevance floor is relaxed to retain at least that many papers by combined score. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `relevance_scoring.min_rank_score` | Minimum composite rank score | +| `relevance_scoring.min_embedding_similarity` | Embedding floor | +| `relevance_scoring.adaptive_embedding` | Adjust floor based on corpus distribution | +| `relevance_scoring.min_keep_papers` | Minimum papers to retain | +| `relevance_scoring.min_concept_match_ratio` | Required concept coverage | + +## LLM + +No. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On failure, returns prior `data` unchanged. + +## Metrics + +Filter counts and adaptive floor adjustments recorded in warnings. + +## Related + +- [Previous: ranking](ranking.md) +- [Next: clustering](clustering.md) +- [Data model: RankedPaper](../data-model.md) diff --git a/docs/architecture/stages/report-generation.md b/docs/architecture/stages/report-generation.md new file mode 100644 index 0000000..c87c339 --- /dev/null +++ b/docs/architecture/stages/report-generation.md @@ -0,0 +1,75 @@ +# Stage: report_generation + +Assembles the final `EnhancedResearchReport` from all pipeline artifacts. + +| | | +|---|---| +| **Class** | `ReportGenerationStage` | +| **Module** | `src/reporting/report_generation.py` | +| **Registry key** | `report_generation` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `dict[str, str]` | Citation exports from citation_export | +| Input (artifacts) | `synthesis_result`, `gap_analysis`, `paper_clusters`, `paper_analyses`, `ranked_papers`, `citation_index` | Full report assembly | +| Output (`data`) | `EnhancedResearchReport` | Final pipeline output | +| Artifacts written | `enhanced_report` | Exported in `ResearchPipelineResult` | + +## Behavior + +`assemble_report()` combines artifacts into a structured report: + +| Report field | Source | +|--------------|--------| +| `executive_summary` | Built from synthesis + embedding-filtered top papers | +| `papers` | `paper_analyses` artifact | +| `clusters` | `paper_clusters` artifact | +| `synthesis` | `synthesis_result` artifact | +| `gap_analysis` | `gap_analysis` artifact | +| `gaps`, `timeline` | Derived from synthesis and paper years | +| `citation_index` | From artifact or rebuilt | +| `exports` | Citation exports from data chain | + +Executive summary uses `relevance_scoring.min_embedding_similarity` as an embedding floor for selecting top papers to highlight. + +Deterministic assembly — no LLM calls. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `relevance_scoring.min_embedding_similarity` | Executive summary embedding floor | + +## LLM + +No — deterministic assembly. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On failure, returns prior `data` unchanged. + +## Metrics + +- `paper_count` +- `cluster_count` + +## Downstream rendering + +The CLI and API render `EnhancedResearchReport` via: + +- `render_enhanced_markdown()` — default markdown output +- `render_report_output()` — markdown, JSON, HTML, PDF-ready formats + +See [Output formats](../../user-guide/output-formats.md). + +## Related + +- [Previous: citation_export](citation-export.md) +- [Data model: EnhancedResearchReport](../data-model.md) +- [Architecture overview](../overview.md) diff --git a/docs/architecture/stages/retrieval.md b/docs/architecture/stages/retrieval.md new file mode 100644 index 0000000..15c015b --- /dev/null +++ b/docs/architecture/stages/retrieval.md @@ -0,0 +1,63 @@ +# Stage: retrieval + +Searches enabled scholarly providers concurrently and merges results. + +| | | +|---|---| +| **Class** | `RetrievalStage` | +| **Module** | `src/retrieval/retrieval_stage.py` | +| **Registry key** | `retrieval` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `ExpandedQuerySet` | Original + variants + sub-questions searched | +| Input (artifact) | `cached_papers` | Session cache bypass — skips provider calls | +| Output (`data`) | `list[RetrievedPaper]` | Merged, deduplicated at provider level | +| Artifacts written | `retrieved_papers` | Used by synthesis recovery | + +## Behavior + +1. If `cached_papers` artifact exists, validates and returns cached papers immediately (cache hit). +2. Otherwise, builds query list from `ExpandedQuerySet` (original + variants + sub-questions). +3. For each enabled provider in config, searches all queries concurrently (bounded by `concurrency_limit`). +4. Provider failures are collected as warnings; partial results continue. +5. All provider results are merged into a single list. + +!!! warning "Per-provider limit caveat" + The stage always passes `settings.retrieval.per_provider_limit` to `provider.search()`. Per-provider `limit` values in YAML are **ignored** at the stage level. + +## Configuration + +| Key | Purpose | +|-----|---------| +| `retrieval.concurrency_limit` | Max concurrent provider requests | +| `retrieval.per_provider_limit` | Papers per provider per query | +| `retrieval.providers.{name}.enabled` | Enable/disable each provider | + +See [Provider matrix](../../retrieval/provider-matrix.md) for live vs stub providers. + +## LLM + +No. + +## Timeout + +`pipeline.stage_timeout_seconds` (default 300 s). + +## Recovery + +On total failure, returns empty list with `partial=True` and warning. + +## Metrics + +- `papers_found` +- `providers_failed` +- `cache_hit` (when applicable) + +## Related + +- [Previous: query_expansion](query-expansion.md) +- [Next: deduplication](deduplication.md) +- [Retrieval overview](../../retrieval/overview.md) diff --git a/docs/architecture/stages/synthesis.md b/docs/architecture/stages/synthesis.md new file mode 100644 index 0000000..fc8ee58 --- /dev/null +++ b/docs/architecture/stages/synthesis.md @@ -0,0 +1,76 @@ +# Stage: synthesis + +Runs two-pass cross-paper synthesis: per-paper extraction followed by collective synthesis. + +| | | +|---|---| +| **Class** | `SynthesisStage` | +| **Module** | `src/analysis/synthesis.py` | +| **Registry key** | `synthesis` | + +## Input / output + +| Direction | Type | Details | +|-----------|------|---------| +| Input (`data`) | `list[PaperCluster]` | From clustering | +| Input (artifacts) | `ranked_papers` | Primary paper source; falls back to `retrieved_papers` | +| Output (`data`) | `SynthesisResult` | Passed to gap_analysis | +| Artifacts written | `paper_extractions`, `paper_analyses`, `synthesis_result`; may refresh `ranked_papers` on recovery | + +## Behavior + +### LLM mode (when `synthesis.llm_enabled`) + +Two-pass flow capped at `max_llm_papers`: + +1. **Pass A — Extraction** (`AgentRole.EXTRACTION`): concurrent per-paper structured extraction (methodology, datasets, benchmarks, limitations, findings) +2. **Pass B — Synthesis** (`AgentRole.SYNTHESIS`): collective cross-paper synthesis JSON (agreements, disagreements, trends, gaps, datasets, methodologies) + +Circuit breaker and retry logic protect against cascading LLM failures. + +### Heuristic mode (default for small Ollama models) + +Extracts key points from abstracts and titles without LLM calls. Produces placeholder text such as *"Details inferred from abstract only"* in downstream report sections. + +### Recovery paths + +- If `ranked_papers` artifact is empty, attempts recovery from `retrieved_papers` via `ensure_ranked_papers()` +- On timeout/cancellation/exception: `recover_synthesis_output()` produces heuristic partial output +- Pipeline-level timeout triggers `recover_stage_output()` in `src/core/stage_recovery.py` + +## Configuration + +| Key | Purpose | +|-----|---------| +| `synthesis.llm_enabled` | Resolved at pipeline start | +| `synthesis.llm_mode` | `auto` / `on` / `off` | +| `synthesis.max_llm_papers` | Cap on LLM extraction calls | +| `synthesis.concurrency` | Parallel extraction limit | +| `synthesis.circuit_breaker_failures` | Failures before circuit opens | +| `llm.*` | Provider/model for agents | + +Env overrides: `RA_SYNTHESIS__LLM_ENABLED`, `RA_SYNTHESIS__LLM_MODE`. + +## LLM + +Two-pass when enabled: `AgentRole.EXTRACTION` → `AgentRole.SYNTHESIS`. Heuristic fallback when disabled or on failure. + +## Timeout + +**`pipeline.synthesis_timeout_seconds`** (default **600 s**) — longer than all other stages. + +## Recovery + +Dedicated heuristic recovery via `recover_synthesis_output()` and `src/core/stage_recovery.py`. + +## Metrics + +Synthesis mode, paper count, extraction count recorded in stage metrics and logs. + +## Related + +- [Previous: clustering](clustering.md) +- [Next: gap_analysis](gap-analysis.md) +- [LLM layer](../llm-layer.md) +- [Heuristic vs LLM](../../llm/heuristic-vs-llm.md) +- [Quality: known issues](../../quality/known-issues.md) diff --git a/docs/configuration/environment-variables.md b/docs/configuration/environment-variables.md new file mode 100644 index 0000000..c0d5aef --- /dev/null +++ b/docs/configuration/environment-variables.md @@ -0,0 +1,170 @@ +# Environment Variables + +Authoritative reference for configuration loaded by `AppSettings` (`src/config/settings.py`) and provider-specific env vars read at HTTP call time. + +Copy [`.env.example`](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/.env.example) to `.env` for local overrides. See [Configuration precedence](precedence.md) for merge order. + +## Naming convention + +- Prefix: `RA_` +- Nested settings: double underscore `__` mirrors YAML nesting +- Example: `ranking.top_k` → `RA_RANKING__TOP_K` + +Boolean env values accept standard truthy strings (`true`, `1`, `yes`). + +--- + +## App-wide + +| Variable | Default | Description | +|----------|---------|-------------| +| `RA_CONFIG_DIR` | `config/` (project root) | Override directory for YAML files | +| `RA_DEBUG` | unset | Alias for debug mode (`1`, `true`, `yes`) — OR-combined with `RA_PIPELINE__DEBUG` | + +--- + +## LLM (`RA_LLM__*`) + +| Variable | Default | Description | +|----------|---------|-------------| +| `RA_LLM__PROVIDER` | `ollama` | `ollama`, `openai`, or `anthropic` | +| `RA_LLM__MODEL` | `auto` | Model name; `auto` selects from `config/ollama_models.yaml` (Ollama only) | +| `RA_LLM__BASE_URL` | `http://localhost:11434` | API base URL (Ollama OpenAI-compatible endpoint) | + +!!! info "Ollama base URL (`/v1` suffix)" + `config/default.yaml` uses `http://localhost:11434` (no `/v1`); `.env.example` uses `/v1`. Both are valid — `normalize_openai_base_url()` in `src/models/base.py` appends `/v1` when missing. See also [Configuration precedence](precedence.md#common-pitfalls). +| `RA_LLM__API_KEY` | `None` | Unified API key; checked before provider-specific keys | +| `RA_LLM__TEMPERATURE` | `0.2` | Defined in config; **not currently passed to pydantic-ai models** | +| `RA_LLM__TIMEOUT_SECONDS` | `120` | Defined in config; stage timeouts use pipeline settings instead | + +**Provider-specific key fallbacks** (when `RA_LLM__API_KEY` is unset): + +| Variable | Provider | +|----------|----------| +| `OPENAI_API_KEY` | OpenAI | +| `ANTHROPIC_API_KEY` | Anthropic | +| `OLLAMA_API_KEY` | Ollama (placeholder; server ignores it) | + +--- + +## Synthesis & query expansion + +| Variable | Default | Description | +|----------|---------|-------------| +| `RA_SYNTHESIS__LLM_ENABLED` | `false` | Force LLM synthesis on/off (overrides `llm_mode`) | +| `RA_SYNTHESIS__LLM_MODE` | `auto` | `auto` \| `on` \| `off` — resolved at pipeline start | +| `RA_SYNTHESIS__MAX_LLM_PAPERS` | `3` | Max papers sent to LLM synthesis | +| `RA_SYNTHESIS__CONCURRENCY` | `2` | Parallel LLM synthesis workers | +| `RA_QUERY_EXPANSION__LLM_ENABLED` | `false` | Force LLM query expansion on/off | +| `RA_QUERY_EXPANSION__LLM_MODE` | `auto` | `auto` \| `on` \| `off` | +| `RA_QUERY_EXPANSION__MAX_VARIANTS` | `5` | Max expanded search variants | +| `RA_QUERY_EXPANSION__MAX_SUB_QUESTIONS` | `3` | Max sub-questions from query understanding | + +!!! tip "Quality vs speed" + Defaults keep synthesis and query expansion **heuristic** (`llm_enabled: false`). Enable LLM features for 8B+ local models or cloud providers. See [Heuristic vs LLM](../llm/heuristic-vs-llm.md). + +--- + +## Ranking (`RA_RANKING__*`) + +| Variable | Default | Description | +|----------|---------|-------------| +| `RA_RANKING__TOP_K` | `25` | Papers kept after ranking | +| `RA_RANKING__WEIGHTS__SEMANTIC_RELEVANCE` | `0.20` | Ranking weight | +| `RA_RANKING__WEIGHTS__EMBEDDING_SIMILARITY` | `0.30` | Ranking weight | +| `RA_RANKING__DOMAIN_PENALTY_MULTIPLIER` | `0.5` | Penalty for off-domain papers | +| `RA_RANKING__CANONICAL_BOOST` | `0.0` | Boost for works in `canonical_works.yaml` | + +Full weight list: [YAML reference](yaml-reference.md#ranking). + +--- + +## Retrieval (`RA_RETRIEVAL__*`) + +| Variable | Default | Description | +|----------|---------|-------------| +| `RA_RETRIEVAL__CONCURRENCY_LIMIT` | `4` | Max parallel searches across query variants | +| `RA_RETRIEVAL__PER_PROVIDER_LIMIT` | `8` | Results per provider per query variant (**actual search limit**) | +| `RA_RETRIEVAL__PROVIDERS____ENABLED` | see below | Enable/disable a provider | +| `RA_RETRIEVAL__PROVIDERS____LIMIT` | `8` | **Ignored** at runtime — use `PER_PROVIDER_LIMIT` | + +**Default provider toggles:** + +| Provider | Default enabled | +|----------|-----------------| +| `openalex` | `true` | +| `semantic_scholar` | `true` | +| `arxiv`, `crossref`, `pubmed`, `core`, `dblp` | `false` | + +**Retrieval API keys (not `RA_`-prefixed):** + +| Variable | Required | Description | +|----------|----------|-------------| +| `S2_API_KEY` | No | Semantic Scholar — higher rate limits when set | +| `RA_CROSSREF_MAILTO` | Recommended | CrossRef polite pool (User-Agent mailto) | +| `CROSSREF_MAILTO` | Recommended | Alias for CrossRef mailto | + +--- + +## Pipeline (`RA_PIPELINE__*`) + +| Variable | Default | Description | +|----------|---------|-------------| +| `RA_PIPELINE__DEBUG` | `false` | Write JSON debug dumps to `logs/debug/` after each run | +| `RA_PIPELINE__STREAM_PROGRESS` | `true` | Live Rich progress on stderr (TTY only) | +| `RA_PIPELINE__CONTINUE_ON_STAGE_FAILURE` | `true` | Heuristic recovery instead of abort | +| `RA_PIPELINE__STAGE_TIMEOUT_SECONDS` | `300` | Per-stage timeout (except synthesis) | +| `RA_PIPELINE__SYNTHESIS_TIMEOUT_SECONDS` | `600` | Synthesis stage timeout | +| `RA_PIPELINE__ENABLED_STAGES__` | `true` | Disable individual pipeline stages | + +Stage names: `query_understanding`, `query_expansion`, `retrieval`, `deduplication`, `ranking`, `relevance_scoring`, `clustering`, `synthesis`, `gap_analysis`, `citation_export`, `report_generation`. See [Stage toggles](stage-toggles.md). + +--- + +## Embedding, deduplication, clustering, relevance, memory + +| Variable | Default | Section | +|----------|---------|---------| +| `RA_EMBEDDING__MODEL` | `BAAI/bge-small-en-v1.5` | Embedding model | +| `RA_EMBEDDING__BATCH_SIZE` | `32` | Embedding batch size | +| `RA_EMBEDDING__CACHE_DIR` | `data/embeddings` | Disk cache for embeddings | +| `RA_DEDUPLICATION__ENABLED` | `true` | Enable dedup stage logic | +| `RA_DEDUPLICATION__EMBEDDING_SIMILARITY_THRESHOLD` | `0.92` | Embedding dedup threshold | +| `RA_CLUSTERING__MIN_CLUSTER_SIZE` | `2` | HDBSCAN min cluster size | +| `RA_CLUSTERING__MAX_MACRO_CLUSTERS` | `4` | Max thematic clusters in report | +| `RA_RELEVANCE_SCORING__MIN_RANK_SCORE` | `0.25` | Minimum rank score to keep | +| `RA_RELEVANCE_SCORING__MIN_PAPERS` | `5` | Minimum papers after filtering | +| `RA_MEMORY__DB_PATH` | `data/research.db` | SQLite session database | +| `RA_MEMORY__CACHE_ENABLED` | `false` | Cache retrieval results across runs | + +--- + +## Quick copy-paste blocks + +**Cloud OpenAI with LLM synthesis:** + +```bash +RA_LLM__PROVIDER=openai +RA_LLM__MODEL=gpt-4o-mini +OPENAI_API_KEY=sk-... +RA_SYNTHESIS__LLM_ENABLED=true +RA_QUERY_EXPANSION__LLM_ENABLED=true +``` + +**Enable arXiv + CrossRef (full pipeline / API only):** + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +RA_CROSSREF_MAILTO=you@example.com +``` + +**Debug mode:** + +```bash +RA_PIPELINE__DEBUG=true +# or +RA_DEBUG=1 +``` + +See also: [Configuration precedence](precedence.md), [YAML reference](yaml-reference.md), [Configuration cookbook](../user-guide/configuration-cookbook.md). diff --git a/docs/configuration/precedence.md b/docs/configuration/precedence.md new file mode 100644 index 0000000..8bf3ec7 --- /dev/null +++ b/docs/configuration/precedence.md @@ -0,0 +1,107 @@ +# Configuration Precedence + +Settings are loaded by `AppSettings` in `src/config/settings.py`. Understanding the merge order helps explain why a YAML value “doesn’t stick” or why `.env.example` behaves differently on a fresh clone. + +## Load order (highest to lowest) + +| Priority | Source | Mechanism | +|----------|--------|-----------| +| 1 (highest) | Constructor kwargs | `AppSettings(retrieval={...})` — used by CLI helper and tests | +| 2 | Process environment | `RA_` prefix, nested keys via `__` (e.g. `RA_LLM__MODEL`) | +| 3 | `.env` file | Same rules as env; loaded via pydantic-settings `env_file` | +| 4 | Merged YAML | `config/default.yaml` + overlays (`models.yaml`, `ranking.yaml`, `providers.yaml`) | +| 5 (lowest) | Pydantic field defaults | Defined on nested models in `settings.py` | + +```mermaid +flowchart LR + Init["Constructor kwargs"] --> Env["RA_* env vars"] + Env --> DotEnv[".env file"] + DotEnv --> YAML["config/*.yaml"] + YAML --> Defaults["Code defaults"] +``` + +Source: `AppSettings.settings_customise_sources()` returns `(init_settings, env_settings, dotenv_settings, YamlSettingsSource)`. + +## YAML merge behavior + +`load_yaml_config()` deep-merges files in this order: + +1. **`default.yaml`** — full settings tree (base) +2. **`models.yaml`** — merged into `llm` +3. **`ranking.yaml`** — merged into `ranking` +4. **`providers.yaml`** — merged into `retrieval` + +Files **not** loaded by `AppSettings`: + +| File | Loaded by | Purpose | +|------|-----------|---------| +| `ollama_models.yaml` | `model_selection.py` | Ollama catalog, RAM/disk hints, synthesis defaults | +| `canonical_works.yaml` | `canonical_works.py` | Optional ranking boost for known works | + +Override the config directory with `RA_CONFIG_DIR=/path/to/config`. + +## Post-load resolution + +At pipeline start, `resolve_effective_settings()` computes runtime values that depend on LLM provider and Ollama model catalog: + +- `synthesis.llm_enabled` +- `query_expansion.llm_enabled` +- Ollama `max_llm_papers` hints from `ollama_models.yaml` + +These can differ from raw YAML/env until the pipeline runs. See [Heuristic vs LLM](../llm/heuristic-vs-llm.md). + +## Alternate loader: `AppSettings.from_yaml()` + +For tests or isolated config directories: + +```python +from src.config.settings import AppSettings + +settings = AppSettings.from_yaml(config_dir=Path("tests/fixtures/config")) +``` + +Precedence: **constructor overrides > process environment > YAML > defaults**. This loader **skips `.env`**, so local developer overrides do not leak into test configs. + +## Common pitfalls + +!!! warning "`.env.example` enables debug by default" + The shipped `.env.example` sets `RA_DEBUG=1`, which turns on debug dumps even when `RA_PIPELINE__DEBUG=false`. Remove or comment it for quiet runs. + +!!! info "Ollama base URL" + Code default is `http://localhost:11434` (no `/v1`). `.env.example` uses `/v1`. Ollama providers normalize via `normalize_openai_base_url()` — both work. + +!!! info "Per-provider `limit` vs `per_provider_limit`" + Each provider has a `limit` field in YAML, but the retrieval stage always uses `retrieval.per_provider_limit`. Per-provider limits in YAML are currently **ignored** at search time. + +## Override examples + +**YAML** (`config/providers.yaml`): + +```yaml +retrieval: + providers: + arxiv: + enabled: true +``` + +**Equivalent env:** + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +``` + +**Programmatic (highest precedence):** + +```python +settings = AppSettings( + retrieval={ + "providers": { + "openalex": {"enabled": True}, + "semantic_scholar": {"enabled": True}, + "arxiv": {"enabled": True}, + } + } +) +``` + +See also: [Environment variables](environment-variables.md), [YAML reference](yaml-reference.md), [Stage toggles](stage-toggles.md). diff --git a/docs/configuration/stage-toggles.md b/docs/configuration/stage-toggles.md new file mode 100644 index 0000000..7b0c0c7 --- /dev/null +++ b/docs/configuration/stage-toggles.md @@ -0,0 +1,98 @@ +# Stage Toggles + +Each pipeline stage can be disabled via `pipeline.enabled_stages` in YAML or matching `RA_PIPELINE__ENABLED_STAGES__*` environment variables. + +Source: `PipelineConfig.enabled_stages` in `src/config/settings.py`; checked in `ResearchPipeline._is_stage_enabled()`. + +## All stages (default: enabled) + +| Stage key | Human label | Typical output | +|-----------|-------------|----------------| +| `query_understanding` | Understanding your question | Parsed intent, concepts, sub-questions | +| `query_expansion` | Expanding search queries | Additional search variants | +| `retrieval` | Retrieving papers | `RetrievedPaper` list from scholarly APIs | +| `deduplication` | Removing duplicates | Deduplicated paper set | +| `ranking` | Ranking papers | `RankedPaper` list (top-k) | +| `relevance_scoring` | Scoring semantic relevance | Filtered ranked papers | +| `clustering` | Grouping by theme | `PaperCluster` groups | +| `synthesis` | Synthesizing insights | Cross-paper themes and analyses | +| `gap_analysis` | Identifying research gaps | Gap findings | +| `citation_export` | Formatting citations | BibTeX/APA/MLA/Chicago strings | +| `report_generation` | Organizing final report | `EnhancedResearchReport` | + +Stage deep dives: [Pipeline stages](../architecture/pipeline-stages.md). + +## YAML configuration + +```yaml +pipeline: + enabled_stages: + query_understanding: true + query_expansion: true + retrieval: true + deduplication: true + ranking: true + relevance_scoring: true + clustering: true + synthesis: true + gap_analysis: true + citation_export: true + report_generation: true +``` + +**Disable clustering** (faster runs, flat paper list in report): + +```yaml +pipeline: + enabled_stages: + clustering: false +``` + +**Retrieval-only smoke test** (skip analysis and reporting): + +```yaml +pipeline: + enabled_stages: + synthesis: false + gap_analysis: false + citation_export: false + report_generation: false +``` + +!!! warning "Downstream dependencies" + Disabling early stages (e.g. `retrieval`) causes later stages to receive empty or stale data. Recovery heuristics may produce partial reports. Prefer disabling analysis stages for quick retrieval tests. + +## Environment overrides + +Nested env keys mirror YAML: + +```bash +RA_PIPELINE__ENABLED_STAGES__CLUSTERING=false +RA_PIPELINE__ENABLED_STAGES__GAP_ANALYSIS=false +``` + +Disable multiple stages by setting each key independently. + +## Related pipeline settings + +These are **not** stage toggles but affect stage behavior: + +| Setting | Default | Effect | +|---------|---------|--------| +| `pipeline.continue_on_stage_failure` | `true` | On failure/timeout, run heuristic recovery instead of aborting | +| `pipeline.stage_timeout_seconds` | `300` | Timeout for all stages except synthesis | +| `pipeline.synthesis_timeout_seconds` | `600` | Synthesis-specific timeout | +| `deduplication.enabled` | `true` | Dedup logic within the deduplication stage | + +When a stage times out or fails with `continue_on_stage_failure=true`, `recover_stage_output()` supplies fallback data and the run is marked **partial**. Progress output shows a ⚠ icon for partial stages. + +## Timeouts vs toggles + +Disabling a stage skips it entirely — no timeout applies. Enabled stages respect: + +- **Synthesis:** `synthesis_timeout_seconds` (default 600s) +- **All others:** `stage_timeout_seconds` (default 300s) + +Set timeout to `0` to disable the asyncio wait (unlimited; not recommended for retrieval). + +See also: [Configuration precedence](precedence.md), [Logging and debug](../operations/logging-and-debug.md), [Progress streaming](../operations/progress-streaming.md). diff --git a/docs/configuration/yaml-reference.md b/docs/configuration/yaml-reference.md new file mode 100644 index 0000000..cc4bb1e --- /dev/null +++ b/docs/configuration/yaml-reference.md @@ -0,0 +1,285 @@ +# YAML Reference + +YAML files in `config/` provide the baseline settings merged into `AppSettings`. Environment variables and `.env` override these values at runtime. + +See [Configuration precedence](precedence.md) for load order. + +## Files loaded by `AppSettings` + +| File | Merged into | Role | +|------|-------------|------| +| `default.yaml` | entire tree | Base defaults for all sections | +| `models.yaml` | `llm` | LLM provider overrides | +| `ranking.yaml` | `ranking` | Ranking weights and top-k | +| `providers.yaml` | `retrieval` | Provider toggles and retrieval limits | + +## Files loaded separately + +| File | Loader | Role | +|------|--------|------| +| `ollama_models.yaml` | `model_selection.py` | Supported Ollama models, RAM/disk requirements, synthesis hints | +| `canonical_works.yaml` | `canonical_works.py` | Optional DOI/title boosts during ranking | + +--- + +## `default.yaml` — full tree + +The base file mirrors all nested models. Key sections: + +### LLM + +```yaml +llm: + provider: ollama + model: auto + base_url: http://localhost:11434 + temperature: 0.2 + timeout_seconds: 120 +``` + +Env override: `RA_LLM__PROVIDER=openai`, `RA_LLM__MODEL=gpt-4o-mini`. + +### Embedding + +```yaml +embedding: + model: BAAI/bge-small-en-v1.5 + batch_size: 32 + cache_dir: data/embeddings +``` + +Used by deduplication, ranking, relevance scoring, and clustering stages. + +### Ranking + +```yaml +ranking: + top_k: 25 + weights: + semantic_relevance: 0.20 + citation_count: 0.08 + recency: 0.08 + venue_quality: 0.10 + abstract_completeness: 0.10 + keyword_overlap: 0.10 + author_prominence: 0.05 + embedding_similarity: 0.30 + domain_penalty_multiplier: 0.5 + outlier_embedding_gap: 0.12 + keyword_collision_max_sim: 0.40 + canonical_boost: 0.0 +``` + +`ranking.yaml` overlays only this section — edit weights without touching `default.yaml`. + +### Query expansion + +```yaml +query_expansion: + llm_mode: auto # auto | on | off + max_variants: 5 + max_sub_questions: 3 +``` + +`llm_enabled` defaults to `false` in code; resolved at pipeline start from `llm_mode` and Ollama catalog. + +### Deduplication + +```yaml +deduplication: + enabled: true + enable_embedding_dedup: true + embedding_similarity_threshold: 0.92 +``` + +### Clustering + +```yaml +clustering: + min_cluster_size: 2 + min_samples: 1 + noise_merge_threshold: 0.5 + max_macro_clusters: 4 +``` + +### Relevance scoring + +```yaml +relevance_scoring: + min_rank_score: 0.25 + min_embedding_similarity: 0.35 + require_all_concepts: true + min_papers: 5 + concept_match_mode: any_group + adaptive_embedding: true + keep_percentile: 25 + gap_from_top: 0.12 +``` + +### Retrieval + +```yaml +retrieval: + concurrency_limit: 4 + per_provider_limit: 8 + providers: + openalex: + enabled: true + limit: 8 + semantic_scholar: + enabled: true + limit: 8 + arxiv: + enabled: false + crossref: + enabled: false + pubmed: + enabled: false + core: + enabled: false + dblp: + enabled: false +``` + +!!! info "Provider `limit` field" + The retrieval stage uses `per_provider_limit` for all providers. Individual `providers..limit` values are **not** applied during search. + +### Pipeline + +```yaml +pipeline: + continue_on_stage_failure: true + stage_timeout_seconds: 300 + synthesis_timeout_seconds: 600 + stream_progress: true + debug: false + enabled_stages: + query_understanding: true + query_expansion: true + retrieval: true + deduplication: true + ranking: true + relevance_scoring: true + clustering: true + synthesis: true + gap_analysis: true + citation_export: true + report_generation: true +``` + +### Memory + +```yaml +memory: + db_path: data/research.db + cache_enabled: false +``` + +Set `cache_enabled: true` to reuse cached retrieval results keyed by query + enabled providers + config hash. + +### Synthesis + +```yaml +synthesis: + llm_mode: auto + max_llm_papers: 3 + extraction_max_retries: 0 + collective_max_retries: 0 + concurrency: 2 + circuit_breaker_failures: 2 +``` + +--- + +## `models.yaml` + +Thin overlay for LLM settings: + +```yaml +provider: ollama +model: auto +base_url: http://localhost:11434 +temperature: 0.2 +timeout_seconds: 120 +``` + +Equivalent env block: + +```bash +RA_LLM__PROVIDER=ollama +RA_LLM__MODEL=auto +RA_LLM__BASE_URL=http://localhost:11434 +``` + +--- + +## `providers.yaml` + +Retrieval-only overlay — same structure as the `retrieval:` section in `default.yaml`. Use this file to enable optional providers without editing the base config: + +```yaml +providers: + arxiv: + enabled: true + crossref: + enabled: true +``` + +Equivalent env: + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +``` + +--- + +## `ollama_models.yaml` + +Not merged into `AppSettings`. Consumed by setup and model auto-selection: + +```yaml +auto_select: true +fallback: llama3.2:3b + +models: + - name: llama3.1:8b + label: Llama 3.1 8B + min_ram_gb: 8 + recommended_ram_gb: 10 + disk_gb: 5 + priority: 100 + synthesis: + llm_enabled: true + max_llm_papers: 5 + + - name: llama3.2:3b + label: Llama 3.2 3B + min_ram_gb: 4 + recommended_ram_gb: 6 + disk_gb: 2.5 + priority: 50 + synthesis: + llm_enabled: false + max_llm_papers: 3 +``` + +When `RA_LLM__MODEL=auto`, setup picks the highest-priority model whose RAM/disk requirements fit the machine. See [Ollama](../llm/ollama.md). + +--- + +## `canonical_works.yaml` + +Optional ranking boost for well-known works: + +```yaml +works: + - title: "Attention Is All You Need" + authors: ["Vaswani"] + year: 2017 + doi_prefix: "10.48550/arXiv.1706.03762" +``` + +Controlled by `ranking.canonical_boost` (default `0.0` — no boost until raised). + +See also: [Environment variables](environment-variables.md), [Stage toggles](stage-toggles.md), [Retrieval overview](../retrieval/overview.md). diff --git a/docs/contributing.md b/docs/contributing.md new file mode 100644 index 0000000..4a2fdee --- /dev/null +++ b/docs/contributing.md @@ -0,0 +1,7 @@ +# Contributing + +The canonical contributor guide lives at the repository root: + +**[contributing.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md)** + +That file covers code and documentation contributions, local MkDocs preview, and writing conventions. Do not duplicate it here — edit the root file so GitHub and the docs site stay in sync. diff --git a/docs/development/extensibility.md b/docs/development/extensibility.md new file mode 100644 index 0000000..c6ae492 --- /dev/null +++ b/docs/development/extensibility.md @@ -0,0 +1,240 @@ +# Extensibility + +The research pipeline is composable through registries for retrieval providers, LLM providers, pipeline stages, and full-text extensions. Built-ins register at startup via `bootstrap_default_plugins()`. + +Source: `src/core/registry.py`, `src/retrieval/providers/registry.py`, `src/models/factory.py`, `tests/test_phase3_extensibility.py`. + +## Architecture + +```mermaid +flowchart TD + bootstrap[bootstrap_default_plugins] --> reg[PluginRegistry] + reg --> providers[Retrieval providers] + reg --> stages[Pipeline stages] + reg --> fulltext[Full-text stubs] + + providers --> retrieval[RetrievalStage] + stages --> pipeline[ResearchPipeline] + pipeline --> execute[execute query] +``` + +Two registries cooperate: + +| Registry | Module | Contents | +|----------|--------|----------| +| `PluginRegistry` | `src/core/registry.py` | Providers, stages, full-text downloaders/indexes | +| Retrieval provider map | `src/retrieval/providers/registry.py` | Provider class lookup for search orchestration | +| LLM provider map | `src/models/factory.py` | Chat model backends | + +`register_retrieval_provider()` updates **both** the retrieval registry and `PluginRegistry`. + +## Bootstrap defaults + +`bootstrap_default_plugins()` registers: + +**Retrieval providers (7):** + +| Name | Status | +|------|--------| +| `openalex`, `semantic_scholar`, `arxiv`, `crossref` | Implemented | +| `pubmed`, `core`, `dblp` | Stub — normalize works; search raises `NotImplementedError` | + +**Pipeline stages (11):** + +`query_understanding` → `query_expansion` → `retrieval` → `deduplication` → `ranking` → `relevance_scoring` → `clustering` → `synthesis` → `gap_analysis` → `citation_export` → `report_generation` + +**Full-text scaffolds:** `stub` PDF downloader and RAG index (both raise `NotImplementedError`). + +The API `/health` endpoint lists registered providers and stages from this registry. + +## Custom retrieval provider + +Implement `RetrievalProvider` in `src/retrieval/providers/base.py`: + +```python +from __future__ import annotations + +import aiohttp +from src.retrieval.models import RetrievedPaper +from src.retrieval.providers.base import RetrievalProvider +from src.retrieval.providers.registry import register_provider +from src.core.registry import register_retrieval_provider + +class ExampleProvider(RetrievalProvider): + name = "example" + + async def search( + self, + session: aiohttp.ClientSession, + query: str, + limit: int | None = None, + ) -> list[RetrievedPaper]: + # Call external API, return normalized papers + ... + + def normalize(self, raw: dict) -> RetrievedPaper: + return RetrievedPaper( + title=raw["title"], + abstract=raw.get("abstract", ""), + year=raw.get("year"), + provider=self.name, + ... + ) + +# Register at import time (or in bootstrap) +register_provider(ExampleProvider) +register_retrieval_provider(ExampleProvider) +``` + +Enable in config: + +```yaml +retrieval: + providers: + example: + enabled: true +``` + +```bash +RA_RETRIEVAL__PROVIDERS__EXAMPLE__ENABLED=true +``` + +Required methods: `search`, `normalize`. Optional override: `health_check`, `_ping`. + +## Custom LLM provider + +Implement `LLMProvider` and register with `register_llm_provider()`: + +```python +from pydantic_ai.models import Model +from src.models.base import LLMProvider +from src.models.factory import register_llm_provider + +class MyLLMProvider(LLMProvider): + name = "my_llm" + + def create_model(self, config) -> Model: + # Return pydantic-ai Model instance + ... + +register_llm_provider(MyLLMProvider) +``` + +Set `RA_LLM__PROVIDER=my_llm`. See [Cloud providers](../llm/cloud-providers.md) for built-in examples. + +## Custom pipeline stage + +Stages implement an async `run(ctx, data) -> StageResult` contract. Register a factory: + +```python +from src.core.registry import register_stage +from src.core.context import PipelineContext, StageResult + +class MyStage: + name = "my_stage" + + async def run(self, ctx: PipelineContext, data): + return StageResult(output=data, duration_ms=1.0) + +register_stage("my_stage", MyStage) +``` + +Wire into a custom pipeline in application code: + +```python +from src.core.pipeline import ResearchPipeline +from src.core.registry import get_registry +from src.config.settings import AppSettings + +settings = AppSettings() +registry = get_registry() +stages = [ + registry.create_stage("query_understanding"), + registry.create_stage("my_stage"), + registry.create_stage("report_generation"), +] +pipeline = ResearchPipeline(stages, settings) +``` + +Built-in stage order and artifacts: [Pipeline stages](../architecture/pipeline-stages.md). + +## Pipeline events + +Subscribe to stage lifecycle events with `StageEventCollector` (`src/core/events.py`): + +```python +from src.core.events import StageEventCollector +from src.core.pipeline import ResearchPipeline + +collector = StageEventCollector() +bus = collector.attach() + +pipeline = ResearchPipeline(stages, settings, event_bus=bus) +await pipeline.execute("query") + +print(collector.started) # [(stage_name, ctx), ...] +print(collector.completed) # [(stage_name, result), ...] +``` + +Useful for metrics exporters, custom progress UI, or audit logging. + +## Full-text extensions (scaffold) + +Registry slots exist for future PDF download and RAG indexing: + +| Slot | Default | Status | +|------|---------|--------| +| `fulltext_downloaders` | `stub` → `StubPDFDownloader` | `NotImplementedError` | +| `fulltext_indexes` | `stub` → `StubRAGIndex` | `NotImplementedError` | + +Register replacements: + +```python +from src.core.registry import get_registry + +registry = get_registry() +registry.register_fulltext_downloader("my_downloader", MyDownloaderFactory) +registry.register_fulltext_index("my_index", MyIndexFactory) +``` + +## API integration + +`create_app()` calls `bootstrap_default_plugins()` before mounting routes. Custom plugins must register **before** the first request (import side effect or custom bootstrap wrapper): + +```python +from my_plugins import register_all +from src.api.app import create_app + +register_all() +app = create_app() +``` + +## Stub provider pattern + +PubMed, CORE, and DBLP demonstrate the intended stub lifecycle: + +1. **Register** in bootstrap (visible in `/health`) +2. **Disable by default** in `config/default.yaml` +3. **Implement `normalize()`** for test fixtures and future API mapping +4. **Raise `NotImplementedError`** in `search()` until live integration ships +5. **Health check** reports unavailable with explanatory message + +Tests in `TestPhase3ProviderStubs` lock this behavior. + +## Testing extensions + +| Test class | Validates | +|------------|-----------| +| `TestPluginRegistry` | Bootstrap registers providers + stages | +| `TestPhase3ProviderStubs` | Stub search raises, normalize maps fields | +| `TestPipelineEvents` | Event collector receives start/complete | +| `TestApiScaffold` | FastAPI factory with plugins loaded | + +Run: `pipenv run pytest tests/test_phase3_extensibility.py -v` + +## Related pages + +- [Retrieval overview](../retrieval/overview.md) — provider orchestration +- [Pipeline stages](../architecture/pipeline-stages.md) — stage artifacts and order +- [Testing](testing.md) — extensibility test patterns +- [Provider matrix](../retrieval/provider-matrix.md) — live vs stub providers diff --git a/docs/development/import-conventions.md b/docs/development/import-conventions.md new file mode 100644 index 0000000..d7e01d1 --- /dev/null +++ b/docs/development/import-conventions.md @@ -0,0 +1,129 @@ +# Import Conventions + +The codebase uses **relative imports inside `src/`** and **absolute `src.*` imports from tests and external scripts**. The `setups/` package sits beside `src/` and uses absolute imports into `src`. + +## Rules + +| Context | Import style | Example | +|---------|--------------|---------| +| Inside `src/` package | Relative (`from .` / `from ..`) | `from ..models import AgentFactory` | +| Tests (`tests/`) | Absolute `src.*` | `from src.config.settings import AppSettings` | +| External scripts / REPL | Absolute `src.*` | `from src.retrieval.orchestrator import build_pipeline` | +| `setups/` modules | Absolute `src.*` | `from src.config.model_selection import resolve_target_model` | +| Type-checking only | `TYPE_CHECKING` guard | Avoid circular imports at runtime | + +## Inside the package + +Modules under `src/` import siblings and parents with relative paths: + +```python +# src/retrieval/providers/openalex.py +from ..models import RetrievedPaper + +# src/analysis/synthesis.py +from ..models import AgentFactory, AgentRole +from ..retrieval.models import RankedPaper, SynthesisResult +from ..core.context import PipelineContext, StageResult +``` + +**Depth guide:** + +| From | To | Prefix | +|------|-----|--------| +| `src/retrieval/providers/foo.py` | `src/retrieval/models.py` | `from ..models` | +| `src/analysis/synthesis.py` | `src/models/` | `from ..models` | +| `src/core/pipeline.py` | `src/config/settings.py` | `from ..config.settings` | + +Prefer relative imports in new `src/` code to keep packages relocatable. + +## Lazy imports + +Some modules defer imports to break cycles or avoid heavy optional deps: + +```python +# src/models/factory.py — defers model_selection import +def _resolve_config(config): + from ..config.model_selection import resolve_llm_model_name + ... + +# src/api/app.py — FastAPI only when create_app() runs +def create_app(...): + from fastapi import FastAPI + ... +``` + +Follow this pattern when adding optional dependencies — do not import FastAPI at module top level in core pipeline code. + +## TYPE_CHECKING blocks + +Use for type hints that would create circular imports: + +```python +from __future__ import annotations +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from ..config.settings import LLMConfig +``` + +Runtime code uses string annotations or deferred imports inside functions. + +## Tests + +All test files use absolute imports from the repo root: + +```python +from src.config.settings import AppSettings +from src.core.pipeline import ResearchPipeline +from src.research.query_expansion import expand_query_heuristic +``` + +Run pytest from the **repository root** so `src` resolves. Commands: [Testing](testing.md). + +Do not use relative imports in `tests/`. + +## External scripts and notebooks + +```python +from src.retrieval.orchestrator import run_research_helper, build_pipeline +from src.models import AgentFactory, create_llm_provider +from src.config.settings import AppSettings, get_settings + +from setups import run_setup, print_report +``` + +Ensure the project root is on `PYTHONPATH` (Pipenv shell handles this automatically). + +## Setups package + +`setups/` is not under `src/` but imports from it: + +```python +# setups/ollama.py +from src.config.model_selection import resolve_target_model +from src.utils.logging_system import logger +``` + +Some setup scripts add the parent directory to `sys.path` for CLI invocation — prefer `pipenv run python -m setups.ollama` over ad-hoc path hacks. + +## Package entry points + +| Entry | Module | +|-------|--------| +| CLI | `python -m src` → `src/__main__.py` | +| API | `uvicorn src.api.app:create_app --factory` | +| Setup | `python -m setups.manager` | + +## Anti-patterns + +| Avoid | Why | +|-------|-----| +| `from src.foo import bar` inside `src/` | Breaks package-relative layout convention | +| Inline imports in hot paths | Reserved for cycle breaking and optional deps only | +| Importing test helpers from `src/` | Keep test utilities in `tests/helpers/` | + +## Related pages + +- [Local development setup](local-setup.md) — run commands and IDE setup +- [Extensibility](extensibility.md) — where to register plugins +- [Architecture overview](../architecture/overview.md) — module boundaries diff --git a/docs/development/local-setup.md b/docs/development/local-setup.md new file mode 100644 index 0000000..ef2135b --- /dev/null +++ b/docs/development/local-setup.md @@ -0,0 +1,120 @@ +# Local Development Setup + +Development environment for AI Research Assistant: Python 3.13, Pipenv, pytest, and optional docs tooling. + +Source: `Pipfile`, `README.md`, `src/__main__.py`. + +## Prerequisites + +| Requirement | Notes | +|-------------|-------| +| Python 3.13+ | Pinned in `Pipfile` `[requires]` | +| Pipenv | Virtualenv and lockfile management | +| Git | Clone the repository | + +For LLM development with Ollama, run [Setup system — Quick start](../setup-system/index.md#quick-start). Cloud-only development needs API keys only. + +## Initial setup + +```bash +git clone https://github.com/Ndevu12/Research_Assistant_Model.git +cd Research_Assistant_Model +pip install pipenv +pipenv install --dev +cp .env.example .env # optional +``` + +`pipenv install --dev` installs: + +- **Runtime packages** (`[packages]`): pydantic-ai, aiohttp, sentence-transformers, hdbscan, etc. +- **Dev packages** (`[dev-packages]`): mkdocs, mkdocs-material, mkdocs-mermaid2-plugin + +Optional API development: + +```bash +pipenv install fastapi uvicorn +``` + +## Run the application + +Always use Pipenv so dependencies resolve correctly. Copy-paste invocations: [CLI reference](../user-guide/cli.md) and [README Usage](https://github.com/Ndevu12/Research_Assistant_Model#usage). + +!!! warning "Plain python may miss deps" + Running `python -m src` outside the Pipenv shell can fail on imports like `sentence-transformers`. Use `pipenv run` or `pipenv shell` first. + +## Shell workflow + +```bash +pipenv shell +python -m src "your query" +``` + +From the shell, run `pytest` and `mkdocs serve` per [Testing](testing.md) and [Publishing docs](publishing.md). + +Exit the shell with `exit` or Ctrl+D. + +## Configuration for development + +| Task | Approach | +|------|----------| +| Persistent local overrides | Edit `.env` (see [Environment variables](../configuration/environment-variables.md)) | +| YAML experiments | Add overlay files under `config/` or set `RA_CONFIG_DIR` | +| Debug pipeline dumps | `RA_PIPELINE__DEBUG=true` or `RA_DEBUG=1` → `logs/debug/` | +| Fast iteration (heuristic) | Default 3B Ollama or `RA_SYNTHESIS__LLM_ENABLED=false` | +| LLM integration testing | Pin `llama3.1:8b` or use cloud keys — all unit tests mock LLM by default | + +Comment out `RA_DEBUG=1` in `.env.example` unless you want debug JSON on every run. + +## Project layout (development) + +``` +Research_Assistant_Model/ +├── src/ # Application package (python -m src) +│ ├── __main__.py # CLI entry +│ ├── api/ # Optional FastAPI layer +│ ├── config/ # Settings, LLM resolution +│ ├── core/ # Pipeline, registry, context +│ ├── research/ # Query expansion, ranking, clustering +│ ├── retrieval/ # Providers, retrieval stage +│ ├── analysis/ # Synthesis, gap analysis +│ └── reporting/ # Report generation, exports +├── tests/ # pytest suite +├── config/ # YAML defaults +├── setups/ # Ollama install, health check +├── docs/ # MkDocs source +└── logs/ # Runtime logs (gitignored) +``` + +## Common development tasks + +| Task | Command / link | +|------|----------------| +| Run all tests | [Testing](testing.md) | +| Skip slow subprocess tests | [Testing — Run tests](testing.md#run-tests) | +| Single test file | [Testing — Run tests](testing.md#run-tests) | +| Docs preview | [Publishing docs](publishing.md) | +| Strict docs build | [Publishing docs](publishing.md) | +| Health check | [Setup system](../setup-system/index.md#quick-start) | +| API server | [API overview](../api/index.md#install-and-run) | + +## IDE / editor notes + +- Set the Python interpreter to the Pipenv virtualenv: `pipenv --venv` +- Mark `src/` as sources root if your IDE supports it +- Tests import via `from src....` — run pytest from repo root + +## Troubleshooting + +| Issue | Fix | +|-------|-----| +| `ModuleNotFoundError: sentence_transformers` | `pipenv install` inside project root | +| Ollama connection errors | [Setup system](../setup-system/index.md#quick-start) | +| Tests hang on subprocess | Use `pytest -m "not slow"` | +| MkDocs strict warnings | Fix broken nav links; run `mkdocs build --strict` locally | + +## Related pages + +- [Testing](testing.md) — test map and mocking strategy +- [Import conventions](import-conventions.md) — relative vs absolute imports +- [Installation](../getting-started/installation.md) — end-user install guide +- [Publishing docs](publishing.md) — GitHub Pages deploy diff --git a/docs/development/publishing.md b/docs/development/publishing.md new file mode 100644 index 0000000..7c99909 --- /dev/null +++ b/docs/development/publishing.md @@ -0,0 +1,91 @@ +# Publishing Documentation + +The documentation site is built with [MkDocs Material](https://squidfunk.github.io/mkdocs-material/) and deployed to GitHub Pages by the [Deploy docs](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/.github/workflows/docs.yml) workflow. + +**Published URL:** https://ndevu12.github.io/Research_Assistant_Model/ + +## One-time repository setup + +Before the first deploy succeeds, configure GitHub Pages in the repository settings: + +1. Open **Settings → Pages**. +2. Set **Build and deployment → Source** to **GitHub Actions** (not “Deploy from a branch”). +3. Merge documentation changes into `main`; the workflow deploys only from that branch. + +After the first successful run, the site is available at the URL above. Optionally: + +- Add the docs URL to the repository **About** section (Website field). +- Add a docs badge or link in the root [README](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/README.md). + +## When deployments run + +The workflow triggers on: + +- **Push to `main`** when files under `docs/`, `mkdocs.yml`, or `.github/workflows/docs.yml` change. +- **Manual dispatch** via **Actions → Deploy docs → Run workflow**. + +Feature branches do not deploy. Validate changes locally or open a PR — the docs workflow runs the same checks without deploying. + +## Local preview and build + +Install dev dependencies and serve the site locally: + +```bash +pipenv install --dev +pipenv run mkdocs serve +# → http://127.0.0.1:8000/Research_Assistant_Model/ +``` + +Build without serving (same check CI uses): + +```bash +NO_MKDOCS_2_WARNING=1 pipenv run mkdocs build --strict +pipenv run python scripts/check_docs_policy.py +``` + +The `--strict` flag turns warnings into errors. Fix any reported issues before pushing doc changes. + +Material for MkDocs prints an informational banner about upcoming MkDocs 2.0 incompatibilities. This project stays on MkDocs 1.x for now; set `NO_MKDOCS_2_WARNING=1` (as in CI) to suppress that banner during local builds. + +## Pull request validation + +The [Deploy docs](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/.github/workflows/docs.yml) workflow also runs on **pull requests** that touch `docs/`, `mkdocs.yml`, or `contributing.md`. PR jobs build but do **not** deploy. + +| Check | Purpose | +|-------|---------| +| `scripts/check_docs_policy.py` | Rejects scaffold placeholders, duplicate `docs/contributing.md`, missing canonical markers, and forbidden duplicate command blocks (see [Canonical sources](../reference/canonical-sources.md)) | +| `mkdocs build --strict` | Fails on broken nav, missing pages, or MkDocs warnings | +| `linkchecker` | Validates internal links in the built `site/` | + +Feature branches do not deploy. Merge to `main` to publish. + +## CI workflow overview + +```mermaid +flowchart LR + pr["Pull request\n(docs paths)"] --> validate["build job\npolicy + strict + links"] + push["Push to main\n(docs paths)"] --> validate + validate --> artifact["upload-pages-artifact\n(main only)"] + artifact --> deploy["deploy job\ndeploy-pages"] + deploy --> site["GitHub Pages site"] +``` + +| Job | Purpose | +|-----|---------| +| `build` | Policy check, `mkdocs build --strict`, link validation; uploads artifact on `main` only | +| `deploy` | Publishes the artifact to the `github-pages` environment (push to `main` only) | + +Build dependencies are installed with `pip` in CI (not Pipenv) for a fast, docs-only job. Local development still uses the versions pinned in `Pipfile` under `[dev-packages]`. + +## Troubleshooting deploys + +| Symptom | Likely cause | Fix | +|---------|--------------|-----| +| Workflow does not run | Change was not on `main` or outside watched paths | Merge to `main` or use workflow dispatch | +| `build` fails on `--strict` | Broken nav link, missing page, or MkDocs warning | Run `pipenv run mkdocs build --strict` locally and fix warnings | +| `build` fails on policy check | Scaffold marker, thin/duplicate page, or forbidden command duplication | Run `pipenv run python scripts/check_docs_policy.py`; see [Canonical sources](../reference/canonical-sources.md) | +| `build` fails on link check | Broken internal doc link | Rebuild and run `linkchecker site/index.html` locally | +| `deploy` fails or site is 404 | Pages source not set to GitHub Actions | Check **Settings → Pages → Source** | +| Broken assets or wrong base URL | `site_url` mismatch | Ensure `site_url` in `mkdocs.yml` ends with `/Research_Assistant_Model/` | + +See also: [Contributing to Documentation](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md) for writing conventions and [Local setup](local-setup.md) for the full dev environment. diff --git a/docs/development/testing.md b/docs/development/testing.md new file mode 100644 index 0000000..bca3865 --- /dev/null +++ b/docs/development/testing.md @@ -0,0 +1,163 @@ +# Testing + + + +pytest suite covering configuration, pipeline stages, retrieval providers, LLM layer, CLI, and extensibility. All tests run offline with mocks — no live LLM or scholarly API calls in CI. + +Source: `tests/test_*.py` (28 files). Internal index: [`docs/_analysis/test-behavior-index.md`](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/docs/_analysis/test-behavior-index.md) (repo-only; not published on the docs site). + +## Run tests + +```bash +pipenv install --dev +pipenv run pytest # full suite +pipenv run pytest -v # verbose +pipenv run pytest -m "not slow" # skip subprocess integration tests +pipenv run pytest tests/test_synthesis.py -v # single file +``` + +Async tests use `pytest-asyncio` (configured for auto mode on async test functions). + +## Test map by domain + +| Domain | Files | What they verify | +|--------|-------|------------------| +| Config / LLM resolution | `test_config_settings.py`, `test_resolve_llm_features.py`, `test_model_selection.py` | YAML merge, env overrides, auto LLM flags, Ollama catalog selection | +| Pipeline core | `test_pipeline_core.py`, `test_paper_adapters.py` | Stage ordering, partial failure, disabled stages, metrics | +| Research stages | `test_research_stages.py`, `test_research_quality.py` | Expansion, ranking, relevance, clustering, dedup; multi-domain quality | +| Retrieval | `test_retrieval_stage.py`, `test_providers.py` | Provider failure tolerance, normalization, health checks | +| Synthesis / gaps | `test_synthesis.py` | Heuristic + LLM paths, stage recovery, timeout handling | +| Reporting | `test_reporting.py`, `test_export.py` | Markdown/JSON/HTML, citations, executive summary | +| LLM providers | `test_llm_providers.py`, `test_graceful_response_handling.py` | Provider registry, base URL normalization, JSON retry/fallback utils | +| CLI / interactive | `test_main_mode_detection.py`, `test_interactive_mode.py`, `test_complete_workflow.py`, others | Mode detection, session UX, subprocess flows | +| Memory / filters | `test_memory.py`, `test_interactive_filters.py` | SQLite sessions, follow-up filters | +| Extensibility | `test_phase3_extensibility.py` | Registry bootstrap, stub providers, API scaffold, events | +| Progress | `test_progress_reporter.py` | TTY detection, stage labels | + +## Mocking strategy + +### LLM calls + +All LLM integration tests **mock pydantic-ai** — no Ollama or cloud API required: + +| Pattern | Example location | +|---------|------------------| +| `patch create_llm_agent` | `test_synthesis.py` | +| `patch` OpenAI/Pydantic AI constructors | `test_llm_providers.py` | +| `MagicMock(EnhancedResponseHandler)` | synthesis workflow tests | + +This keeps CI fast and deterministic. Manual LLM verification uses the CLI with real providers. + +### Retrieval providers + +| Pattern | Purpose | +|---------|---------| +| `SuccessProvider` / `FailingProvider` / `EmptyProvider` stubs | Stage-level retrieval tests | +| `patch get_enabled_providers` | Control which providers run | +| `AsyncMock` aiohttp sessions | Provider health check tests | +| Normalization unit tests | Raw API payload → `RetrievedPaper` mapping | + +### Embeddings + +| Fixture | Purpose | +|---------|---------| +| `FixedEmbeddingProvider` | Deterministic vectors for ranking/relevance/quality tests | +| `MockEmbeddingProvider` | Lightweight stub for stage tests | +| `patch.object(provider, "_load_model")` | Skip sentence-transformers model load | + +### Pipeline stubs + +| Stub | Purpose | +|------|---------| +| `EchoStage`, `FailingStage`, `PartialStage` | Pipeline core behavior | +| `RetrievalStub` | End-to-end stage chain without HTTP | +| `mock_pipeline_result` (`tests/helpers/pipeline_mocks.py`) | Orchestrator output tests | + +### CLI / subprocess + +| Pattern | Notes | +|---------|-------| +| `patch sys.argv` + `patch asyncio.run` | Unit-test `__main__` without subprocess | +| `@pytest.mark.slow` subprocess tests | `test_complete_workflow.py` — real `python -m src` | +| `capsys` | Assert stdout/stderr formatting | + +Skip slow tests in quick loops: `pytest -m "not slow"`. + +## Key test behaviors + +### LLM feature resolution (`test_resolve_llm_features.py`) + +| Test | Confirms | +|------|----------| +| `test_llm_mode_auto_8b` | Ollama 8B + auto → LLM on, `max_llm_papers=5` | +| `test_llm_mode_auto_3b` | Ollama 3B + auto → LLM off | +| `test_cloud_provider_auto_enables_llm` | OpenAI + auto → LLM on | +| `test_env_llm_enabled_overrides_mode` | Env bool beats `llm_mode: off` | + +### Multi-domain quality (`test_research_quality.py`) + +Parametrized cases across NLP, biomedical, climate, and economics domains: + +- No degenerate query variants +- Embedding outlier demotes homonym decoys +- Adaptive relevance filter drops off-topic papers +- Executive summary excludes decoy terms +- No hardcoded ML-specific branch constants in source + +### Extensibility (`test_phase3_extensibility.py`) + +- Stub providers (PubMed, CORE, DBLP) registered but `NotImplementedError` on search +- `bootstrap_default_plugins()` registers 7 providers + 11 stages +- `StageEventCollector` fires start/complete events +- FastAPI `create_app` requires optional dependency + +## Fixtures and helpers + +| Path | Role | +|------|------| +| `tests/helpers/pipeline_mocks.py` | `mock_pipeline_result()` for orchestrator tests | +| `catalog_dir` fixture | Temp `ollama_models.yaml` for selection tests | +| `temp_config_dir` fixture | YAML overlay merge tests | +| `memory_store` fixture | Tmp SQLite for session tests | + +## Coverage gaps + +Document these when adding tests: + +| Gap | Detail | +|-----|--------| +| Query understanding | No dedicated unit test file | +| API routes | Scaffold tests only — no HTTP integration tests | +| `EnhancedResponseHandler` | Subcomponents tested; not end-to-end | +| Live LLM / API | All mocked in unit tests | +| Subprocess tests | Marked `@pytest.mark.slow`; may skip in tight CI | + +## Writing new tests + +1. **Import style**: use absolute imports (`from src.module import ...`) in tests — see [Import conventions](import-conventions.md). +2. **Async stages**: mark with `@pytest.mark.asyncio`. +3. **Avoid live network**: mock aiohttp or patch provider classes. +4. **Deterministic embeddings**: prefer `FixedEmbeddingProvider` over real sentence-transformers loads. +5. **Config isolation**: use `AppSettings(...)` kwargs or `monkeypatch` for env — do not rely on developer `.env`. + +Example minimal stage test: + +```python +import pytest +from src.config.settings import AppSettings +from src.core.context import PipelineContext +from src.research.query_expansion import QueryExpansionStage + +@pytest.mark.asyncio +async def test_expansion_produces_variants() -> None: + stage = QueryExpansionStage() + ctx = PipelineContext(settings=AppSettings(), query="machine learning") + result = await stage.run(ctx, "machine learning") + assert len(result.output.variants) >= 1 +``` + +## Related pages + +- [Local development setup](local-setup.md) — install and run commands +- [Extensibility](extensibility.md) — registry patterns tested in phase3 +- [Heuristic vs LLM](../llm/heuristic-vs-llm.md) — behavior under test in synthesis/resolve tests diff --git a/docs/getting-started/health-check.md b/docs/getting-started/health-check.md new file mode 100644 index 0000000..8a6a7f0 --- /dev/null +++ b/docs/getting-started/health-check.md @@ -0,0 +1,77 @@ +# Health Check + +## Commands + +Setup commands (health check, manager, model pin): [Setup system — Quick start](../setup-system/index.md#quick-start). + +Copy-paste install steps: [README Setup & health check](https://github.com/Ndevu12/Research_Assistant_Model#setup--health-check). + +To override the model for a one-off check, see [Setup system — Module reference](../setup-system/index.md#health_checkpy) (`--model` flag). + +## What `health_check` validates + +Source: `setups/health_check.py` + +| Check | Function | Pass criteria | +|-------|----------|---------------| +| Pipenv available | `check_python_deps()` | `pipenv` on PATH, `aiohttp` importable | +| Embedding deps | `check_embedding_deps()` | `sentence-transformers` importable | +| Ollama binary | `check_ollama_installed()` | `ollama` on PATH | +| Ollama server | `check_ollama_running()` | `ollama list` succeeds (5s timeout) | +| Model | `check_model_available()` | Resolved model exists locally | + +Model resolution uses `resolve_target_model()` from `src/config/model_selection.py`: + +- Reads `RA_LLM__MODEL` (default `auto`) +- When `auto`, picks best fit from `config/ollama_models.yaml` based on RAM/disk +- Validates against installed Ollama models via `ollama list` + +Override the resolved model for a check: see [Setup system](../setup-system/index.md#health_checkpy) (`health_check --model …`). + +## Integration with CLI startup + +Every `python -m src` invocation runs `ensure_setup()` first (`src/__main__.py`): + +1. If provider ≠ `ollama` → skip setup, return success +2. Run `health_check.check_ollama_running()` + `check_model_available()` +3. If both pass → proceed +4. Else → print health report → `manager.run_setup()` → abort if setup fails + +This is why first-run can take several minutes (Ollama install + model pull). + +```mermaid +flowchart TD + CLI["python -m src"] --> ES[ensure_setup] + ES --> P{provider == ollama?} + P -->|no| OK[Continue to pipeline] + P -->|yes| HC[health_check] + HC --> R{ollama OK and model OK?} + R -->|yes| OK + R -->|no| MGR[manager.run_setup] + MGR --> S{success?} + S -->|yes| OK + S -->|no| EXIT[exit 1] +``` + +## Reading the report + +`health_check.print_report()` prints a human-readable status for each check. Overall success requires embedding deps + (for Ollama) running server + installed model. + +Common failure messages: + +| Message | Action | +|---------|--------| +| `sentence-transformers not installed` | `pipenv install` | +| `Ollama server is not running` | [Setup system](../setup-system/index.md#quick-start) or `ollama serve` | +| `Model '…' not installed` | [Setup system](../setup-system/index.md#quick-start) or `ollama pull ` | +| `pipenv not installed` | Install Pipenv, run from project root | + +## Cloud provider note + +When `RA_LLM__PROVIDER` is `openai` or `anthropic`, CLI setup checks are skipped. Set provider and API keys per [Cloud providers](../llm/cloud-providers.md), then run a query using [README Usage](https://github.com/Ndevu12/Research_Assistant_Model#usage). + +## Troubleshooting + +See [Troubleshooting](../operations/troubleshooting.md) for expanded FAQ. + +See also: [Installation](installation.md), [Quick start](quick-start.md), [Ollama](../llm/ollama.md). diff --git a/docs/getting-started/installation.md b/docs/getting-started/installation.md new file mode 100644 index 0000000..9d14ea8 --- /dev/null +++ b/docs/getting-started/installation.md @@ -0,0 +1,77 @@ +# Installation + +## Commands + +Copy-paste install steps: [README Requirements and Quick Start](https://github.com/Ndevu12/Research_Assistant_Model#requirements). + +This page covers what gets installed, layout, and verification — not repeated bash blocks from the README. + +## What you need + +| Requirement | Version / notes | +|-------------|-----------------| +| Python | 3.13+ | +| Pipenv | Dependency and virtualenv management | +| Internet | API retrieval; optional after Ollama model download for local LLM | +| RAM (local LLM) | 4–6 GB (`llama3.2:3b`) or 8–10 GB (`llama3.1:8b`) | + +Cloud LLM providers (OpenAI, Anthropic) need only an API key — no Ollama install. + +!!! warning "Default quality profile" + Out of the box, synthesis and query expansion run in **heuristic mode** (`llm_enabled: false`). Reports are fast but use template-driven cross-paper analysis rather than LLM-authored synthesis. Enable `RA_SYNTHESIS__LLM_ENABLED=true` and `RA_QUERY_EXPANSION__LLM_ENABLED=true` for higher quality with 8B+ local or cloud models. See [Heuristic vs LLM](../llm/heuristic-vs-llm.md). + +## What gets installed + +`pipenv install` reads `Pipfile` / `Pipfile.lock` and installs: + +| Package category | Examples | Role | +|------------------|----------|------| +| LLM agents | `pydantic-ai` | Structured LLM calls | +| HTTP | `aiohttp` | Scholarly API retrieval | +| Embeddings | `sentence-transformers` | Dedup, ranking, clustering | +| Config | `pydantic`, `pydantic-settings` | Settings and schemas | +| Clustering | `hdbscan` | Thematic paper groups | +| CLI UX | `rich` | Progress streaming | + +**Not included by default:** FastAPI and uvicorn (optional API layer). See [API overview](../api/index.md). + +## Project layout (install-relevant) + +``` +Research_Assistant_Model/ +├── config/ # YAML defaults (merged at runtime) +├── src/ # Application code (`python -m src`) +├── setups/ # Ollama install, health check, model pull +├── Pipfile # Dependency manifest +├── .env.example # Template for local secrets +└── data/ # Created at runtime (embeddings cache, SQLite) +``` + +`data/`, `logs/`, and `reports/` are gitignored and created on first run. + +## Optional `.env` setup + +Copy `.env.example` to `.env` before first run if you want persistent overrides: + +```bash +cp .env.example .env +``` + +!!! warning "Debug flag in example file" + `.env.example` sets `RA_DEBUG=1`. Comment it out unless you want debug JSON dumps in `logs/debug/` on every run. + +Key sections in `.env`: LLM provider, retrieval API keys, pipeline flags. Full reference: [Environment variables](../configuration/environment-variables.md). + +## Verify installation + +Run the health check flow: [Health check](health-check.md) (commands live in [Setup system](../setup-system/index.md#quick-start)). + +## Development install + +For tests and docs tooling, install dev dependencies per [Local development setup](../development/local-setup.md#initial-setup), then run tests via [Testing](../development/testing.md). + +## Next steps + +- [Quick start](quick-start.md) — first query and what runs internally +- [Health check](health-check.md) — validate Ollama and models +- [Configuration precedence](../configuration/precedence.md) diff --git a/docs/getting-started/quick-start.md b/docs/getting-started/quick-start.md new file mode 100644 index 0000000..d77b0aa --- /dev/null +++ b/docs/getting-started/quick-start.md @@ -0,0 +1,96 @@ +# Quick Start + +## Commands + +Install and run your first query: [README Quick Start](https://github.com/Ndevu12/Research_Assistant_Model#quick-start). + +CLI flags and output formats: [README Usage](https://github.com/Ndevu12/Research_Assistant_Model#usage) and [CLI reference](../user-guide/cli.md). + +Always use **`pipenv run`** — plain `python -m src` may miss dependencies (see [Troubleshooting](../operations/troubleshooting.md)). + +## What runs internally + +### 1. CLI entry (`src/__main__.py`) + +``` +python -m src "query" + → main() + → ensure_setup() + → run_research_helper(query) +``` + +### 2. Setup gate (`ensure_setup()`) + +When `RA_LLM__PROVIDER=ollama` (default): + +1. Warn if not inside Pipenv (`PIPENV_ACTIVE != 1`) +2. Import `setups/health_check.py` and `setups/manager.py` +3. Check Ollama server + resolved model via `health_check` +4. If incomplete → print report → run `manager.run_setup()` (install/start Ollama, pull model) + +Cloud providers (`openai`, `anthropic`) **skip** Ollama setup entirely. + +```mermaid +flowchart TD + Start["python -m src"] --> Setup["ensure_setup()"] + Setup --> Provider{LLM provider?} + Provider -->|ollama| HC[health_check] + HC --> OK{Ollama + model OK?} + OK -->|no| MGR[manager.run_setup] + OK -->|yes| Run[run_research_helper] + MGR --> Run + Provider -->|openai/anthropic| Run +``` + +### 3. Research helper (`run_research_helper`) + +Batch CLI queries use a **shortcut settings override**: + +- Providers: **OpenAlex + Semantic Scholar only** (YAML toggles ignored) +- Limit: 8 papers per provider per query variant (default) +- Pipeline: full 11 stages via `build_pipeline()` → `ResearchPipeline.execute()` + +Interactive mode (`python -m src` with no query) uses `InteractiveResearchSession` → `run_research_with_result()` with **full** `AppSettings` (all enabled providers from config). + +### 4. Pipeline execution + +``` +run_research_with_result() + → resolve_effective_settings() # LLM feature flags + → PipelineProgressReporter # stderr progress (TTY) + → build_pipeline() # 11 stages + → pipeline.execute(query) + → render_report_output() # markdown/json/html/pdf +``` + +Stages run sequentially: query understanding → expansion → retrieval → dedup → ranking → relevance → clustering → synthesis → gap analysis → citations → report. + +See [Architecture overview](../architecture/overview.md). + +## First-run timeline (Ollama default) + +| Step | What happens | +|------|----------------| +| 1 | Pipenv deps already installed | +| 2 | Ollama installed/started if missing | +| 3 | Model resolved (`auto` → `llama3.1:8b` or `llama3.2:3b` from catalog) | +| 4 | Model pulled if not local | +| 5 | Embedding model downloaded on first embedding stage | +| 6 | Scholarly APIs queried; report printed to stdout | + +Progress appears on stderr unless `--no-progress` or non-TTY. + +## Default quality profile + +Out of the box, the pipeline runs **heuristic synthesis** and **heuristic query expansion** (`llm_enabled: false`). Reports are fast and work offline after model download, but cross-paper synthesis is template-driven rather than LLM-authored. + +Enable LLM stages and copy-paste env recipes: [Heuristic vs LLM](../llm/heuristic-vs-llm.md) and [Configuration cookbook](../user-guide/configuration-cookbook.md). + +## After your first query + +| Goal | Next page | +|------|-----------| +| Validate setup | [Health check](health-check.md) | +| CLI flags and formats | [CLI reference](../user-guide/cli.md) | +| Enable arXiv / CrossRef | [Configuration cookbook](../user-guide/configuration-cookbook.md) | +| Debug a partial run | [Logging and debug](../operations/logging-and-debug.md) | diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 0000000..4b872a0 --- /dev/null +++ b/docs/index.md @@ -0,0 +1,25 @@ +# AI Research Assistant + +A local-first research pipeline that retrieves academic papers from multiple scholarly APIs, ranks and clusters them with embeddings, synthesizes cross-paper insights, and exports reports in several formats. Uses **Ollama** by default for fully local LLM inference, with optional **OpenAI** and **Anthropic** providers. + +Built with Python 3.13, pydantic-ai, sentence-transformers, and async I/O. + +Feature list, requirements, and RAM guidance: [README](https://github.com/Ndevu12/Research_Assistant_Model#features). + +!!! warning "Default quality profile" + Synthesis and query expansion default to **heuristic mode** (`llm_enabled: false`). Reports are fast but template-driven. Enable LLM features for richer cross-paper analysis — see [Heuristic vs LLM](llm/heuristic-vs-llm.md). + +## Quick links + +| Section | Description | +|---------|-------------| +| [Installation](getting-started/installation.md) | Pipenv, Python 3.13, dependencies | +| [Quick Start](getting-started/quick-start.md) | First query and auto-setup flow | +| [CLI Reference](user-guide/cli.md) | Flags, batch vs interactive mode | +| [CLI vs API](user-guide/cli-vs-api.md) | Execution paths and provider divergence | +| [Configuration](configuration/precedence.md) | Env vars, YAML, and precedence | +| [Known Issues](quality/known-issues.md) | Research quality analysis and fix backlog | +| [Architecture](architecture/overview.md) | End-to-end pipeline overview | +| [Canonical sources](reference/canonical-sources.md) | Where copy-paste commands live (link, don't duplicate) | + +Install and run: [README Quick Start](https://github.com/Ndevu12/Research_Assistant_Model#quick-start). diff --git a/docs/llm/cloud-providers.md b/docs/llm/cloud-providers.md new file mode 100644 index 0000000..e130c32 --- /dev/null +++ b/docs/llm/cloud-providers.md @@ -0,0 +1,169 @@ +# Cloud Providers + +OpenAI and Anthropic are supported as cloud LLM backends. Both use pydantic-ai model wrappers and share the same `AgentFactory` / role system as Ollama. + +Source: `src/models/openai.py`, `src/models/anthropic.py`, `src/models/factory.py`. + +## When to use cloud providers + +| Scenario | Recommendation | +|----------|----------------| +| No local GPU/RAM for 8B+ models | OpenAI (`gpt-4o-mini`) or Anthropic (`claude-3-5-haiku-latest`) | +| Maximum synthesis quality | Cloud + explicit `RA_SYNTHESIS__LLM_ENABLED=true` | +| Offline / privacy-sensitive work | Stay on [Ollama](ollama.md) | +| Cost-sensitive batch runs | Heuristic mode or smaller cloud models | + +!!! tip "Auto-enables LLM stages" + With `llm_mode: auto`, **OpenAI and Anthropic always enable** synthesis and query expansion LLM — unlike Ollama, which follows catalog hints. See [Heuristic vs LLM](heuristic-vs-llm.md). + +## OpenAI + +### Configuration + +```bash +RA_LLM__PROVIDER=openai +RA_LLM__MODEL=gpt-4o-mini +OPENAI_API_KEY=sk-... +``` + +**YAML equivalent:** + +```yaml +llm: + provider: openai + model: gpt-4o-mini + # base_url optional — defaults to https://api.openai.com +``` + +| Variable | Required | Notes | +|----------|----------|-------| +| `OPENAI_API_KEY` | **Yes** | Primary key source | +| `RA_LLM__API_KEY` | No | Overrides provider-specific key when set | +| `RA_LLM__BASE_URL` | No | OpenAI-compatible proxy (LM Studio, Azure OpenAI-style endpoints) | + +### Base URL behavior + +`OpenAIProviderImpl` (`src/models/openai.py`): + +- Empty base URL or Ollama default (`http://localhost:11434`) → `https://api.openai.com/v1` +- Custom base URL → normalized with `/v1` suffix + +Use a custom base URL for OpenAI-compatible gateways: + +```bash +RA_LLM__PROVIDER=openai +RA_LLM__MODEL=your-model-name +RA_LLM__BASE_URL=https://your-gateway.example.com +RA_LLM__API_KEY=your-key +``` + +Missing API key raises at model creation: + +``` +OpenAI provider requires an API key. Set RA_LLM__API_KEY or OPENAI_API_KEY. +``` + +## Anthropic + +### Configuration + +```bash +RA_LLM__PROVIDER=anthropic +RA_LLM__MODEL=claude-3-5-haiku-latest +ANTHROPIC_API_KEY=sk-ant-... +``` + +**YAML equivalent:** + +```yaml +llm: + provider: anthropic + model: claude-3-5-haiku-latest +``` + +| Variable | Required | Notes | +|----------|----------|-------| +| `ANTHROPIC_API_KEY` | **Yes** | Primary key source | +| `RA_LLM__API_KEY` | No | Unified override | + +Anthropic uses pydantic-ai's native Anthropic model — no custom base URL normalization beyond what pydantic-ai provides. + +## API key resolution order + +All providers check keys through `resolve_api_key()` in `src/models/base.py`: + +| Priority | Source | +|----------|--------| +| 1 | `RA_LLM__API_KEY` (from settings / env) | +| 2 | Provider env var (`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `OLLAMA_API_KEY`) | +| 3 | Ollama only: default `"ollama"` | + +Full env reference: [Environment variables](../configuration/environment-variables.md). + +## Provider selection flow + +```mermaid +flowchart TD + cfg[llm.provider from settings] --> factory[AgentFactory._resolve_config] + factory --> auto{model == auto?} + auto -->|yes Ollama only| catalog[resolve_llm_model_name] + auto -->|no| name[Use llm.model as-is] + catalog --> create[create_llm_provider] + name --> create + create --> lookup[get_llm_provider_class] + lookup --> ollama[OllamaProvider] + lookup --> openai[OpenAIProviderImpl] + lookup --> anthropic[AnthropicProviderImpl] +``` + +!!! warning "auto model with cloud providers" + `llm.model: auto` resolves via `ollama_models.yaml` — meaningful for Ollama only. Set an explicit cloud model name (`gpt-4o-mini`, `claude-3-5-haiku-latest`, etc.). + +## Registering a custom provider + +Extend the built-in registry for OpenAI-compatible or custom backends: + +```python +from src.models.base import LLMProvider +from src.models.factory import register_llm_provider + +class MyProvider(LLMProvider): + name = "my_provider" + + def create_model(self, config): + ... + +register_llm_provider(MyProvider) +``` + +Then set `RA_LLM__PROVIDER=my_provider`. See [Extensibility](../development/extensibility.md). + +## Example recipes + +**High-quality cloud run:** [Configuration cookbook — Cloud OpenAI](../user-guide/configuration-cookbook.md#3-cloud-openai-quality). + +**Anthropic with explicit LLM modes:** + +```bash +RA_LLM__PROVIDER=anthropic +RA_LLM__MODEL=claude-3-5-haiku-latest +ANTHROPIC_API_KEY=sk-ant-... +RA_SYNTHESIS__LLM_MODE=on +RA_QUERY_EXPANSION__LLM_MODE=on +``` + +More recipes: [Configuration cookbook](../user-guide/configuration-cookbook.md). + +## Known limitations + +| Setting | Status | +|---------|--------| +| `llm.temperature` | Defined in config; **not passed** to pydantic-ai constructors | +| `llm.timeout_seconds` | Defined in config; stage timeouts use pipeline settings | +| `llm.model: auto` | Ollama catalog only — use explicit names for cloud | + +## Related pages + +- [Ollama](ollama.md) — local default provider +- [LLM layer](../architecture/llm-layer.md) — agent roles and call sites +- [Heuristic vs LLM](heuristic-vs-llm.md) — cloud auto-enables LLM stages diff --git a/docs/llm/heuristic-vs-llm.md b/docs/llm/heuristic-vs-llm.md new file mode 100644 index 0000000..2cf0153 --- /dev/null +++ b/docs/llm/heuristic-vs-llm.md @@ -0,0 +1,141 @@ +# Heuristic vs LLM + +The pipeline can run query expansion, synthesis, and gap analysis with **heuristics** (no LLM tokens) or **LLM agents** (structured JSON via pydantic-ai). Default settings favor heuristics for speed and low-resource machines. + +Source: `src/config/resolve_llm_features.py`, `src/research/query_expansion.py`, `src/analysis/synthesis.py`, `src/analysis/gap_analysis.py`. + +## Default behavior + +Out of the box on Ollama with `llama3.2:3b` (catalog fallback) and `llm_mode: auto`: + +| Stage | LLM used? | Mechanism | +|-------|-----------|-----------| +| Query expansion | **No** | Synonym tables, acronym expansion, Jaccard gates | +| Synthesis | **No** | Abstract sentence extraction, embedding-aligned agreements | +| Gap analysis | **No** | Derived from heuristic synthesis fields | + +Reports may contain placeholders such as *"Details inferred from abstract only"* and *"Cross-paper disagreement analysis limited in heuristic mode."* + +!!! warning "Quality impact" + A successful pipeline run with **zero LLM tokens** can still produce misleading executive summaries when retrieval returns off-topic papers. See [Known issues](../quality/known-issues.md) for the full analysis and fixes applied to heuristic paths. + +## Feature flag resolution + +`resolve_effective_settings()` runs once at pipeline start and sets `synthesis.llm_enabled` and `query_expansion.llm_enabled` on the config passed to all stages. + +**Precedence** (per feature — synthesis and query expansion independently): + +1. `RA_{SECTION}__LLM_ENABLED` env (`true`/`false`/`1`/`0`) +2. `llm_mode: on` → enabled; `llm_mode: off` → disabled +3. `llm_mode: auto` → rules below + +### Auto-mode rules + +| Provider | synthesis LLM | query expansion LLM | +|----------|---------------|---------------------| +| `openai`, `anthropic` | Always **on** | Always **on** | +| `ollama` | On if catalog entry has `synthesis.llm_enabled: true` | Same catalog hint | +| Other | **Off** | **Off** | + +For Ollama, the resolved model name (after `auto` selection) determines catalog hints: + +| Model | `llm_mode: auto` | `max_llm_papers` hint | +|-------|------------------|----------------------| +| `llama3.1:8b` | LLM **on** | 5 | +| `llama3.2:3b` | LLM **off** | 3 | + +```bash +# Force LLM on any Ollama model +RA_SYNTHESIS__LLM_ENABLED=true +RA_QUERY_EXPANSION__LLM_ENABLED=true +``` + +## What each path does + +### Query expansion + +**Heuristic** (`expand_query_heuristic`): + +- Domain synonym map and acronym expansion +- Phrase-aware variant generation with Jaccard overlap gate +- Broad-term guard to avoid degenerate single-word variants +- No API calls; deterministic for a given query + +**LLM** (`AgentRole.EXPANSION`): + +- Structured JSON: search variants + sub-questions +- Uses `ctx.config` LLM settings after resolution +- Falls back to heuristics on timeout or parse failure + +### Synthesis (two-pass) + +**Heuristic**: + +- Extracts sentences from abstracts aligned with query concepts +- Aggregates agreements from top-quartile embedding-similar papers +- Limited cross-paper disagreement analysis +- No per-paper deep reading + +**LLM**: + +- **Pass A** (`AgentRole.EXTRACTION`): structured per-paper analysis (up to `max_llm_papers`) +- **Pass B** (`AgentRole.SYNTHESIS`): cross-paper synthesis JSON +- Parallel workers controlled by `synthesis.concurrency` +- Timeout recovery via `src/core/stage_recovery.py` + +### Gap analysis + +!!! info "Coupled to synthesis" + Gap analysis LLM is gated by **`synthesis.llm_enabled`**, not a separate `gap_analysis.llm_mode`. When synthesis LLM is off, gap analysis uses heuristics derived from synthesis output. + +**Heuristic gap analysis**: infers gaps from synthesis agreement/disagreement fields. + +**LLM gap analysis** (`AgentRole.GAP_ANALYSIS`): structured gaps and research opportunities JSON. + +## Quality tradeoffs + +| Dimension | Heuristic | LLM | +|-----------|-----------|-----| +| Speed | ~seconds for synthesis stage | Minutes on local 8B; faster on cloud | +| RAM / cost | Minimal | 8–10 GB RAM (local 8B) or API fees (cloud) | +| Query expansion | Good for common CS/ML terms | Better for niche or interdisciplinary queries | +| Synthesis depth | Abstract snippets only | Per-paper extraction + cross-paper reasoning | +| Disagreement analysis | Placeholder text | LLM compares conflicting claims | +| Failure mode | Can rank/ summarize off-topic papers | Same retrieval issues, but richer analysis when papers are relevant | + +## When to enable LLM + +| Goal | Suggested config | +|------|------------------| +| Fast scan, low RAM | Defaults (3B + heuristic) — [Cookbook recipe 1](../user-guide/configuration-cookbook.md) | +| Local quality | `llama3.1:8b` + catalog auto or `RA_SYNTHESIS__LLM_ENABLED=true` — [Recipe 2](../user-guide/configuration-cookbook.md) | +| Best quality | Cloud OpenAI/Anthropic — [Recipe 3](../user-guide/configuration-cookbook.md) | +| Debug heuristic-only bugs | Keep LLM off; inspect `logs/debug/` pipeline dumps | + +## Configuration reference + +| Variable | Default | Effect | +|----------|---------|--------| +| `RA_SYNTHESIS__LLM_MODE` | `auto` | Tri-state synthesis LLM | +| `RA_SYNTHESIS__LLM_ENABLED` | unset | Force override | +| `RA_SYNTHESIS__MAX_LLM_PAPERS` | `3` | Cap LLM extraction pass | +| `RA_QUERY_EXPANSION__LLM_MODE` | `auto` | Tri-state expansion LLM | +| `RA_QUERY_EXPANSION__LLM_ENABLED` | unset | Force override | + +YAML equivalents under `synthesis:` and `query_expansion:` in `config/default.yaml`. See [Stage toggles](../configuration/stage-toggles.md). + +## Verify LLM is active + +Check pipeline logs or debug JSON in `logs/debug/`: + +- Log line: `Resolved LLM features (provider=..., model=...): synthesis=True, query_expansion=True` +- Metrics: non-zero `llm_tokens_in` / `llm_tokens_out` in pipeline metrics + +Enable debug and inspect dumps: [Logging and debug — Debug walkthrough](../operations/logging-and-debug.md#debug-walkthrough). + +## Related pages + +- [Ollama](ollama.md) — catalog hints for auto mode +- [Cloud providers](cloud-providers.md) — auto-enables LLM +- [Synthesis stage](../architecture/stages/synthesis.md) — two-pass workflow +- [Known issues](../quality/known-issues.md) — heuristic quality RCA diff --git a/docs/llm/ollama.md b/docs/llm/ollama.md new file mode 100644 index 0000000..5df4872 --- /dev/null +++ b/docs/llm/ollama.md @@ -0,0 +1,162 @@ +# Ollama + +Ollama is the **default LLM provider**. The application talks to a local Ollama server through its OpenAI-compatible API (`/v1/chat/completions`) via pydantic-ai. + +Source: `src/models/ollama.py`, `src/config/model_selection.py`, `config/ollama_models.yaml`, `setups/ollama.py`. + +## Quick setup + +Setup commands (manager, health check, model pin): [Setup system — Quick start](../setup-system/index.md#quick-start). + +Check matrix and reading the report: [Health check](../getting-started/health-check.md). + +The setup system reads the same catalog as runtime auto-selection. + +## Configuration + +| Setting | Default | Env override | +|---------|---------|--------------| +| Provider | `ollama` | `RA_LLM__PROVIDER=ollama` | +| Model | `auto` | `RA_LLM__MODEL=llama3.1:8b` | +| Base URL | `http://localhost:11434` | `RA_LLM__BASE_URL` | +| API key | placeholder `"ollama"` | `RA_LLM__API_KEY` or `OLLAMA_API_KEY` | + +**YAML** (`config/default.yaml` or overlay): + +```yaml +llm: + provider: ollama + model: auto + base_url: http://localhost:11434 +``` + +!!! info "Base URL normalization" + `normalize_openai_base_url()` appends `/v1` when missing. Both `http://localhost:11434` and `http://localhost:11434/v1` resolve to the same endpoint. See [Environment variables](../configuration/environment-variables.md). + +## Model catalog (`config/ollama_models.yaml`) + +The catalog drives **setup auto-selection**, **health checks**, and **LLM feature hints** when `llm.model` is `auto`. + +```yaml +auto_select: true +fallback: llama3.2:3b + +models: + - name: llama3.1:8b + label: Llama 3.1 8B + min_ram_gb: 8 + recommended_ram_gb: 10 + disk_gb: 5 + priority: 100 + synthesis: + llm_enabled: true + max_llm_papers: 5 + + - name: llama3.2:3b + label: Llama 3.2 3B + min_ram_gb: 4 + recommended_ram_gb: 6 + disk_gb: 2.5 + priority: 50 + synthesis: + llm_enabled: false + max_llm_papers: 3 +``` + +| Field | Purpose | +|-------|---------| +| `priority` | Higher wins when multiple models fit resources | +| `min_ram_gb` / `disk_gb` | Hard requirements for auto-select | +| `recommended_ram_gb` | Used in setup logging and health-check warnings | +| `synthesis.llm_enabled` | Hint for `llm_mode: auto` (see [Heuristic vs LLM](heuristic-vs-llm.md)) | +| `synthesis.max_llm_papers` | Applied to synthesis config when model resolves | + +Override the catalog directory with `RA_CONFIG_DIR` if you maintain a custom copy. + +## Auto-selection algorithm + +When `llm.model` is `auto` or empty, `resolve_llm_model_name()` in `src/config/model_selection.py`: + +```mermaid +flowchart TD + start[resolve_llm_model_name] --> catalog[Load ollama_models.yaml] + catalog --> autoSelect{auto_select?} + autoSelect -->|false| fallback[Use catalog.fallback] + autoSelect -->|true| resources[Detect RAM, disk, swap pressure] + resources --> pick[Highest-priority model that fits] + pick -->|none fit| fallback + pick --> name[Concrete model name] + fallback --> name +``` + +**Resource detection:** + +| OS | RAM source | Disk | +|----|------------|------| +| Linux | `/proc/meminfo` (MemTotal, MemAvailable, swap) | `shutil.disk_usage` on config dir | +| macOS | `sysctl hw.memsize`, `vm_stat` | Same | +| Other | Conservative fallback values | Same | + +**Swap pressure** can downgrade selection when swap is heavily used. Explicit model names (CLI `--model`, env, or YAML) skip auto-selection entirely. + +**Model source priority** (when resolving target model for setup): + +1. CLI `--model` argument +2. `RA_LLM__MODEL` env var +3. YAML `llm.model` +4. Catalog auto-select (or `fallback`) + +## Setup integration + +`setups/ollama.py` shares `resolve_target_model()` with runtime resolution: + +| Command | Behavior | +|---------|----------| +| `python -m setups.ollama install` | OS-aware Ollama binary install (Linux script, Homebrew, Arch pacman/yay) | +| `python -m setups.ollama setup` | Start server if needed, resolve model, `ollama pull` if missing | +| `python -m setups.manager` | Full pipeline: deps → Ollama install → model setup | + +After setup, if the selected catalog entry has `synthesis.llm_enabled: true`, setup logs a tip to set `RA_SYNTHESIS__LLM_ENABLED=true`. + +## LLM feature resolution (Ollama-specific) + +At pipeline start, `resolve_effective_settings()` resolves the concrete model name, then applies `llm_mode: auto` rules: + +| Resolved model | `llm_mode: auto` → synthesis | `llm_mode: auto` → query expansion | +|----------------|------------------------------|-------------------------------------| +| `llama3.1:8b` | **On** (catalog hint) | **On** (same hint) | +| `llama3.2:3b` | **Off** | **Off** | + +Force behavior regardless of catalog: + +```bash +RA_SYNTHESIS__LLM_ENABLED=true +RA_QUERY_EXPANSION__LLM_ENABLED=true +# or +RA_SYNTHESIS__LLM_MODE=on +RA_QUERY_EXPANSION__LLM_MODE=on +``` + +Env `RA_SYNTHESIS__LLM_ENABLED` / `RA_QUERY_EXPANSION__LLM_ENABLED` override `llm_mode` entirely. + +## Provider implementation + +`OllamaProvider` (`src/models/ollama.py`) wraps pydantic-ai's `OpenAIChatModel` pointed at the normalized base URL. No cloud API key is required; the server ignores the placeholder key. + +Agents are created per role at call time via `AgentFactory` — see [LLM layer](../architecture/llm-layer.md) for roles (EXPANSION, EXTRACTION, SYNTHESIS, GAP_ANALYSIS). + +## Troubleshooting + +| Symptom | Check | +|---------|-------| +| `Connection refused` on LLM calls | `ollama list` — start with `ollama serve` or re-run setup | +| Wrong model selected | [Setup system](../setup-system/index.md#quick-start) — review RAM/disk in health check output | +| LLM stages skipped | 3B fallback + `llm_mode: auto` disables LLM; pin 8B or set `RA_SYNTHESIS__LLM_ENABLED=true` | +| Catalog not found | Ensure `config/ollama_models.yaml` exists or set `RA_CONFIG_DIR` | + +## Related pages + +- [Heuristic vs LLM](heuristic-vs-llm.md) — when catalog hints enable LLM stages +- [Cloud providers](cloud-providers.md) — switch away from Ollama +- [Configuration cookbook](../user-guide/configuration-cookbook.md) — copy-paste recipes +- [LLM layer](../architecture/llm-layer.md) — full resolution pipeline diff --git a/docs/operations/logging-and-debug.md b/docs/operations/logging-and-debug.md new file mode 100644 index 0000000..9055ff1 --- /dev/null +++ b/docs/operations/logging-and-debug.md @@ -0,0 +1,124 @@ +# Logging and Debug + + + +The assistant writes structured logs to `logs/` and optional JSON pipeline dumps when debug mode is enabled. + +Sources: `src/utils/logging_system.py`, `src/core/pipeline.py`, `src/config/settings.py`. + +## Log files + +On startup, `_setup_logger()` creates a dated set of rotating log files under `logs/`: + +| File | Contents | Filter | +|------|----------|--------| +| `combined_YYYYMMDD.log` | All log levels | None | +| `error_YYYYMMDD.log` | Warnings and errors | `level >= WARNING` | +| `performance_YYYYMMDD.log` | Performance metrics | `is_performance=True` | +| `events_YYYYMMDD.log` | Structured events | `is_event=True` | + +Files rotate at **10 MB** with **5** backups. The same messages also go to **stdout** via a console handler. + +### Tail logs during a run + +```bash +tail -f logs/combined_$(date +%Y%m%d).log +tail -f logs/error_$(date +%Y%m%d).log +``` + +### Logger API + +```python +from src.utils.logging_system import logger + +logger.info("Pipeline started") +logger.warning("Provider failed") +logger.log_event("retrieval", "Cache hit", extra_data={"key": "..."}) +logger.log_performance("synthesis", duration_ms=4500, success=True) +``` + +## Enabling debug mode + +Debug is active when **either** flag is set: + +| Variable | Values | Source | +|----------|--------|--------| +| `RA_PIPELINE__DEBUG` | `true` / `false` | `pipeline.debug` | +| `RA_DEBUG` | `1`, `true`, `yes` | Direct env check in `debug_enabled` property | + +```bash +RA_PIPELINE__DEBUG=true +# or +RA_DEBUG=1 +``` + +!!! warning "`.env.example` ships with debug on" + Fresh copies of `.env.example` set `RA_DEBUG=1`. Comment it out for production-like runs unless you want debug dumps every time. + +## Pipeline debug dumps + +When `settings.debug_enabled` is true, after each pipeline run `ResearchPipeline._write_debug_dump()` writes: + +``` +logs/debug/pipeline__.json +``` + +The JSON contains `PipelineContext.to_debug_dict()`: + +- Session ID and query +- Per-stage results (duration, partial flag, warnings, metrics) +- Artifact snapshots (ranked papers, synthesis, gap analysis, etc.) +- Configuration effective at run time + +Use these dumps to inspect why a stage was partial or what papers were retrieved before deduplication. + +### Debug walkthrough + +1. **Enable debug:** + + ```bash + export RA_PIPELINE__DEBUG=true + pipenv run python -m src "your query" + ``` + +2. **Run a query** (interactive or batch). + +3. **Find the dump:** + + ```bash + ls -lt logs/debug/ | head + ``` + +4. **Inspect stage results:** + + ```bash + jq '.stage_results.retrieval' logs/debug/pipeline_*.json + jq '.artifacts.ranked_papers | length' logs/debug/pipeline_*.json + ``` + +5. **Check warnings** for partial stages: + + ```bash + jq '[.stage_results[] | select(.partial)] | keys' logs/debug/pipeline_*.json + ``` + +6. **Correlate with combined log** using timestamps and session ID in log lines. + +## Quality metrics (optional) + +`src/utils/quality_monitor.py` can write additional JSON to `logs/quality_metrics_.json` when quality monitoring is active during LLM stages. + +## What debug does *not* include + +- Raw HTTP response bodies from retrieval providers (only normalized papers in artifacts) +- Full LLM prompts (check synthesis stage metrics/logs) +- Automatic log level change — debug dumps are additive; log level stays at INFO unless you configure logging separately + +## Related settings + +| Setting | Effect | +|---------|--------| +| `RA_PIPELINE__STREAM_PROGRESS` | Live Rich UI on stderr (separate from file logging) | +| `RA_PIPELINE__CONTINUE_ON_STAGE_FAILURE` | Failed stages log errors then recover heuristically | + +See also: [Progress streaming](progress-streaming.md), [Troubleshooting](troubleshooting.md), [Stage toggles](../configuration/stage-toggles.md). diff --git a/docs/operations/progress-streaming.md b/docs/operations/progress-streaming.md new file mode 100644 index 0000000..afcdc4a --- /dev/null +++ b/docs/operations/progress-streaming.md @@ -0,0 +1,83 @@ +# Progress Streaming + +During pipeline runs, the CLI can show live progress on **stderr**: stage checkmarks, sub-activities, and streaming LLM token previews. + +Source: `src/utils/progress_reporter.py`, wired in `run_research_with_result()` (`src/retrieval/orchestrator.py`). + +## What you see + +When stderr is a TTY, `PipelineProgressReporter` renders a Rich live display: + +1. **Query header** — the research question +2. **Progress bar** — stages completed / 11 total +3. **Current stage label** — e.g. "Retrieving papers from scholarly APIs…" +4. **Sub-activity** — e.g. "Analyzing paper 2/5" (cyan italic line) +5. **LLM preview panel** — trailing tokens during agent streaming ("AI generating…") +6. **Per-stage lines** — ✓ or ⚠ with duration in ms when each stage completes +7. **Summary** — `Pipeline complete` or `Pipeline partial` with total time + +### Stage labels + +| Stage key | Display label | +|-----------|---------------| +| `query_understanding` | Understanding your question | +| `query_expansion` | Expanding search queries | +| `retrieval` | Retrieving papers from scholarly APIs | +| `deduplication` | Removing duplicate papers | +| `ranking` | Ranking papers by relevance | +| `relevance_scoring` | Scoring semantic relevance | +| `clustering` | Grouping papers by theme | +| `synthesis` | Synthesizing cross-paper insights | +| `gap_analysis` | Identifying research gaps | +| `citation_export` | Formatting citations | +| `report_generation` | Organizing final report | + +## Enabling and disabling + +| Method | Effect | +|--------|--------| +| Default | Enabled when `RA_PIPELINE__STREAM_PROGRESS=true` (default) | +| CLI flag | `--no-progress` disables for one run | +| Env | `RA_PIPELINE__STREAM_PROGRESS=false` | +| Non-TTY stderr | Automatically disabled (e.g. piped output, CI logs) | + +```bash +# Disable for this run +pipenv run python -m src --no-progress "your query" + +# Disable globally +export RA_PIPELINE__STREAM_PROGRESS=false +``` + +## Event bus integration + +The reporter subscribes to `PipelineEventBus` stage lifecycle events: + +```mermaid +sequenceDiagram + participant Pipe as ResearchPipeline + participant Bus as PipelineEventBus + participant Rep as PipelineProgressReporter + Pipe->>Bus: emit_stage_start + Bus->>Rep: _on_stage_start + Pipe->>Bus: emit_stage_complete + Bus->>Rep: _on_stage_complete +``` + +Stages call `set_activity()` / `set_llm_preview()` during long operations (especially synthesis). LLM calls use `stream_agent_text()` which delegates to `stream_agent_response()` when a reporter is active. + +## Context variable + +The active reporter is stored in a `ContextVar` (`get_progress_reporter()` / `set_progress_reporter()`). Nested async tasks in the same pipeline run share one reporter instance. + +## API and non-CLI runs + +`run_research_with_result(..., stream_progress=True)` accepts an explicit override. FastAPI handlers can pass `stream_progress=false` for silent server-side runs. + +Interactive mode respects the same `--no-progress` flag passed at CLI startup. + +## Partial runs + +A stage marked **partial** (timeout, provider failure, heuristic recovery) shows ⚠ instead of ✓. The final summary line reads `Pipeline partial` in yellow. + +See also: [Logging and debug](logging-and-debug.md), [CLI reference](../user-guide/cli.md), [Architecture overview](../architecture/overview.md). diff --git a/docs/operations/troubleshooting.md b/docs/operations/troubleshooting.md new file mode 100644 index 0000000..84f3ed5 --- /dev/null +++ b/docs/operations/troubleshooting.md @@ -0,0 +1,149 @@ +# Troubleshooting + +Consolidated FAQ for common setup, runtime, and quality issues. Command snippets link to the [root README](https://github.com/Ndevu12/Research_Assistant_Model#quick-start); this page focuses on diagnosis. + +## Setup and dependencies + +### "Not running inside Pipenv" warning + +Always use Pipenv so dependencies (especially `sentence-transformers`) are available. See [README Quick Start](https://github.com/Ndevu12/Research_Assistant_Model#quick-start) and [CLI reference](../user-guide/cli.md). + +Plain `python -m src` may miss packages installed only in the Pipenv virtualenv. + +### Missing `sentence-transformers` / import errors + +Run `pipenv install`, then verify with [Setup system — Quick start](../setup-system/index.md#quick-start). + +Embedding stages require `sentence-transformers` and will fail or degrade without it. + +### Setup directory not found + +The CLI expects `setups/` at the project root. Run commands from the repository root, not from `src/`. + +--- + +## Ollama (default LLM) + +### Ollama not installed or not running + +Use [Setup system — Quick start](../setup-system/index.md#quick-start), or manually: `ollama serve` and `ollama pull llama3.1:8b`. + +First CLI run with `RA_LLM__PROVIDER=ollama` triggers automatic setup via `ensure_setup()` in `src/__main__.py`. + +### Wrong or missing model + +| Symptom | Fix | +|---------|-----| +| Model not installed | Set `RA_LLM__MODEL=auto` or run `ollama pull ` | +| Too slow / OOM | Use `RA_LLM__MODEL=llama3.2:3b` or let `auto` pick a smaller model | +| Wrong model selected | Check `config/ollama_models.yaml` priorities and your RAM | + +Pin a model via [Setup system](../setup-system/index.md#quick-start) (`manager --model …`). + +### Skipping Ollama setup + +Cloud providers skip local setup automatically: + +```bash +RA_LLM__PROVIDER=openai +OPENAI_API_KEY=sk-... +``` + +--- + +## Retrieval + +### No papers returned + +1. Check network connectivity +2. Try a broader query +3. Enable debug and inspect `logs/debug/pipeline_*.json` → `stage_results.retrieval` +4. Look for provider warnings in `logs/combined_*.log` + +### Semantic Scholar rate limits + +Set `S2_API_KEY` in `.env` for higher limits. The client retries on 429 with `Retry-After` backoff. + +### CrossRef empty or slow results + +Set polite-pool mailto: + +```bash +RA_CROSSREF_MAILTO=you@example.com +``` + +### arXiv / CrossRef not used from CLI batch mode + +!!! info "CLI vs full pipeline" + `python -m src "query"` only queries OpenAlex + Semantic Scholar. Use interactive mode, the API, or programmatic `run_research()` to enable other providers. + +### Enabled stub provider (PubMed, CORE, DBLP) + +Enabling a stub logs a warning and returns no papers from that provider. Disable in YAML/env until implemented. + +--- + +## Report quality + +### Report feels shallow or generic + +Default config keeps **heuristic synthesis** (`RA_SYNTHESIS__LLM_ENABLED=false`). Enable LLM stages per [Heuristic vs LLM](../llm/heuristic-vs-llm.md) and [Configuration cookbook](../user-guide/configuration-cookbook.md#2-high-quality-local-8b-llm-synthesis). + +See also [Known issues](../quality/known-issues.md). + +### Partial pipeline / warning icons in progress + +Stages can complete with ⚠ when: + +- A retrieval provider failed (others may still succeed) +- A stage timed out (`stage_timeout_seconds` / `synthesis_timeout_seconds`) +- Heuristic recovery ran (`continue_on_stage_failure=true`) + +Inspect debug dumps or error logs for the specific stage warning. + +--- + +## Output and formats + +### HTML/PDF printed to terminal + +Use `--output` for file output. See [Output formats](../user-guide/output-formats.md) and [README Output Format](https://github.com/Ndevu12/Research_Assistant_Model#output-format). + +`--format pdf` produces print-ready HTML — open in a browser → Print → Save as PDF. + +### Citation exports + +See [Output formats — Citation exports](../user-guide/output-formats.md#citation-exports-export). + +--- + +## Configuration surprises + +### YAML change has no effect + +Check precedence: env vars and `.env` override YAML. Verify with debug dump `config` section or temporarily unset conflicting env keys. + +### Debug always on after copying `.env.example` + +`.env.example` sets `RA_DEBUG=1`. Comment it out or set `RA_DEBUG=0`. + +### `per_provider_limit` vs provider `limit` + +Only `RA_RETRIEVAL__PER_PROVIDER_LIMIT` controls search result counts. Per-provider `limit` in YAML is ignored. + +--- + +## Logs and debug + +```bash +tail -f logs/combined_$(date +%Y%m%d).log +ls logs/debug/ +``` + +Full debug walkthrough (including `RA_PIPELINE__DEBUG` run command): [Logging and debug](logging-and-debug.md). + +## Health check reference + +Check matrix and diagnostics: [Health check](../getting-started/health-check.md). Commands: [Setup system](../setup-system/index.md#quick-start). + +See also: [Known issues](../quality/known-issues.md). diff --git a/docs/quality/known-issues.md b/docs/quality/known-issues.md new file mode 100644 index 0000000..dbd4685 --- /dev/null +++ b/docs/quality/known-issues.md @@ -0,0 +1,217 @@ +# Research Quality — Known Issues + +**Status:** P0–P2 fixes **implemented** (2026-05-22) +**Recorded:** 2026-05-22 +**Triggering run:** Interactive query `transformer attention mechanisms` +**Branch context:** `feat/multi-stage-research-pipeline` + +--- + +## Summary + +Reports could look **factually wrong or off-topic** even when the pipeline completed successfully. For the reference run, this was **not** primarily an Ollama/LLM accuracy failure: **no LLM calls were made** (`llm_tokens_in: 0`, `synthesis.llm_enabled=false`). The pipeline retrieved a mix of relevant and irrelevant papers, ranked some off-topic work highly, and assembled a misleading executive summary using **heuristic synthesis** and **abstract snippet extraction**. + +The fixes below are **query-agnostic** — they use embedding similarity, adaptive corpus-relative thresholds, and generic keyword-collision detection rather than hardcoded NLP/ML domain lists. Multi-domain regression tests live in `tests/test_research_quality.py`. + +--- + +## Reference Run — Observed Symptoms (Pre-Fix) + +| Symptom | Example from output | +|---------|---------------------| +| Executive summary off-topic | Opens with *“Air pollution poses a critical global public health challenge…”* for an NLP/transformer query | +| Irrelevant papers in top results | Cervical cancer (CerviFormer), tea evapotranspiration, Arabic sign language, boring machining (PhyDT) | +| Suspicious canonical paper metadata | *Attention Is All You Need* listed as **2025** with DOI `10.65215/2q58a426` | +| Fragmented clustering | Many themes prefixed `Unclustered: …` (8 thin clusters) | +| Generic analysis placeholders | *“Details inferred from abstract only”*, *“Full disagreement analysis requires LLM synthesis”* | +| No relevance pruning | 25 papers kept, **0 filtered** by relevance stage | +| High retrieval volume | 96 papers retrieved → 76 after dedup → 25 ranked | + +**Pipeline duration:** ~7.4s (11 stages, no failures) +**Debug artifact:** `logs/debug/pipeline_*_20260522_*.json` (query: `transformer attention mechanisms`) +**Ranking top score:** `0.8922` (misleadingly high for topical fit) + +--- + +## Fix Status + +| ID | Component | Status | What changed | +|----|-----------|--------|--------------| +| RC-1 | Query expansion | **Fixed** | Phrase-aware synonyms, Jaccard gate, broad-term guard; ML-specific templates removed | +| RC-2 | Ranking | **Fixed** | Embedding weight 30%; embedding outlier + keyword-collision penalties (no ML branch) | +| RC-3 | Relevance gate | **Fixed** | Adaptive embedding floor (percentile + gap-from-top); configurable `min_embedding_similarity` | +| RC-4 | Synthesis & summary | **Fixed** | Top-quartile agreements; template executive summary wired to config thresholds | +| RC-5 | Clustering | **Fixed** | Macro-cluster merge when HDBSCAN noise > 50%; `Theme:` labels instead of singleton spam | +| RC-6 | Metadata / dedup | **Fixed** | Generic year/DOI sanity; `canonical_boost: 0.0` default (registry opt-in) | +| RC-7 | LLM mode | **Fixed** | `llm_mode: auto \| on \| off` via `resolve_llm_features.py` | + +--- + +## Architecture Context + +``` +Query → expansion (heuristic) → retrieval (OpenAlex + Semantic Scholar) + → dedup → ranking → relevance_scoring → clustering → synthesis (heuristic) + → gap_analysis (heuristic) → report_generation +``` + +| Stage | LLM used? | Quality notes | +|-------|-----------|---------------| +| Query expansion | Auto (off on 3B Ollama) | Heuristic variants with quality gates | +| Retrieval | No | Keyword/API search | +| Ranking | No | Weighted signals + embedding cache | +| Relevance | No | Adaptive embedding floor + concept match | +| Synthesis | Auto | Heuristic path uses embedding-aligned agreements | +| Report | No | Template summary + validated snippet | + +--- + +## Root Cause Analysis (Historical) + +These sections document **why** the reference run failed. Each maps to a fix above. + +### RC-1: Query expansion produces noisy search strings + +**Location:** `src/research/query_expansion.py` + +For query `transformer attention mechanisms`, heuristic expansion previously yielded redundant variants like `attention mechanism attention mechanisms`. Fixes: phrase-aware replacement, Jaccard overlap gate, broad-term guard, concept bigrams. **No** transformer/attention-specific templates. + +--- + +### RC-2: Ranking mis-weighted for topical precision + +**Location:** `src/research/ranking.py`, `config/default.yaml` + +Previously `embedding_similarity` was 5% and `semantic_relevance` was keyword-only. Now embedding weight is 30%, with generic penalties for: + +- Partial core-concept coverage +- Embedding outliers (far below top-5 mean similarity) +- Keyword collision (high overlap, low embedding similarity) + +Config: `outlier_embedding_gap: 0.12`, `keyword_collision_max_sim: 0.40`. + +--- + +### RC-3: Relevance filter was effectively disabled + +**Location:** `src/research/relevance_scoring.py` + +Previously filtered only at `rank_score < 0.05`. Now uses composite gate with adaptive embedding floor: + +```python +floor = max(min_embedding_similarity, percentile(sims, keep_percentile), top_sim - gap_from_top) +``` + +Percentile skipped when corpus size < 8. Default `min_embedding_similarity: 0.35`. + +--- + +### RC-4: Heuristic synthesis built misleading narratives + +**Location:** `src/analysis/synthesis.py`, `src/reporting/report_generation.py` + +Previously `agreements[0]` from rank order drove the executive summary. Now: + +- Agreements drawn from top embedding-similarity quartile +- Executive summary uses query + cluster themes template +- Validated snippet only when embedding similarity ≥ configured floor + +--- + +### RC-5: Clustering fragmented into “Unclustered” singletons + +**Location:** `src/research/clustering.py` + +When HDBSCAN noise ratio > `noise_merge_threshold` (0.5), noise papers merge into ≤4 keyword macro-clusters labeled `Theme: …`. + +--- + +### RC-6: Scholarly metadata quality + +**Location:** `src/research/metadata_sanity.py`, `src/retrieval/deduplication.py` + +Generic rules: future-year correction, DOI format checks, richer duplicate preference. Canonical work registry is **opt-in** (`canonical_boost: 0.0` default). + +--- + +### RC-7: LLM synthesis disabled by default + +**Location:** `src/config/resolve_llm_features.py`, `config/default.yaml` + +Tri-state `llm_mode: auto` enables LLM on cloud providers and capable local models (8B+). Small Ollama models (e.g. `llama3.2:3b`) stay heuristic-only unless explicitly overridden. + +--- + +## Quality Modes + +| Mode | Typical setup | Synthesis | Expansion | +|------|---------------|-----------|-----------| +| Heuristic-only | `llama3.2:3b`, `llm_mode: auto` | Off | Off | +| Balanced local | `llama3.1:8b`, `llm_mode: auto` | On | On | +| Cloud | `openai` / `anthropic`, `llm_mode: auto` | On | On | + +Override with `RA_SYNTHESIS__LLM_MODE=on|off|auto` or legacy `RA_SYNTHESIS__LLM_ENABLED=true|false`. + +--- + +## Reproduction Checklist + +Run multi-domain regression (no network): + +```bash +pipenv run pytest tests/test_research_quality.py -q +``` + +Full pipeline smoke test: + +```bash +pipenv run python -m src "transformer attention mechanisms" +``` + +**Expect (post-fix):** + +- [ ] Executive summary uses query + cluster themes; no off-topic decoy domain in lead +- [ ] Keyword-collision decoys demoted in ranking and filtered by relevance +- [ ] High HDBSCAN noise yields ≤4 `Theme:` macro-clusters, not N singletons +- [ ] `llm_tokens_in: 0` when running heuristic-only mode on 3B model + +**Inspect expansion:** + +```bash +pipenv run python -c " +from src.research.query_expansion import expand_query_heuristic, extract_core_concepts +q = 'transformer attention mechanisms' +print(expand_query_heuristic(q, extract_core_concepts(q))) +" +``` + +--- + +## Related Configuration + +| Setting | Default | Quality impact | +|---------|---------|----------------| +| `synthesis.llm_mode` | `auto` | Model-aware LLM enable (RC-7) | +| `query_expansion.llm_mode` | `auto` | Model-aware expansion (RC-1) | +| `ranking.weights.embedding_similarity` | `0.30` | Primary semantic steering (RC-2) | +| `ranking.outlier_embedding_gap` | `0.12` | Demote embedding outliers (RC-2) | +| `ranking.keyword_collision_max_sim` | `0.40` | Keyword-without-semantics penalty (RC-2) | +| `ranking.canonical_boost` | `0.0` | Opt-in landmark boost (RC-6) | +| `relevance_scoring.min_embedding_similarity` | `0.35` | Base embedding floor (RC-3) | +| `relevance_scoring.adaptive_embedding` | `true` | Corpus-relative floor (RC-3) | +| `clustering.noise_merge_threshold` | `0.5` | Macro-cluster merge trigger (RC-5) | + +Environment overrides: + +- `RA_RELEVANCE_SCORING__MIN_EMBEDDING_SIMILARITY` +- `RA_SYNTHESIS__LLM_MODE=auto` +- `RA_RANKING__CANONICAL_BOOST=0.05` (opt-in) + +--- + +## Change Log + +| Date | Action | +|------|--------| +| 2026-05-22 | Initial documentation from interactive run analysis | +| 2026-05-22 | P0–P2 fixes implemented; multi-domain tests in `test_research_quality.py` | diff --git a/docs/reference/canonical-sources.md b/docs/reference/canonical-sources.md new file mode 100644 index 0000000..08e1996 --- /dev/null +++ b/docs/reference/canonical-sources.md @@ -0,0 +1,95 @@ + + +# Canonical Sources + +Every **copy-paste command block** and **repeated conceptual warning** lives in one canonical place. All other pages **link** to it — like [docs/contributing.md](../contributing.md) does for the root [contributing.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md). + +This page is the registry for contributors. When you add or move documentation, check here first so commands and warnings are not duplicated across nav pages. + +## Domain hubs + +Canonical content is organized by topic hub, not a single mega commands page: + +```mermaid +flowchart TB + subgraph root [Repo root — GitHub entry] + README["README.md\ninstall + first query + usage cheat sheet"] + CONTRIB["contributing.md\ncontributor + mkdocs CI"] + end + subgraph getting [Getting Started] + INSTALL["installation.md\nwhat gets installed — links README"] + QUICK["quick-start.md\ninternals only — links README"] + HEALTH["health-check.md\ncheck matrix — links setup-system"] + end + subgraph hubs [Docs canonical hubs] + SETUP["setup-system/index.md\nhealth_check + manager"] + CLI["user-guide/cli.md\nflags + execution paths"] + FORMATS["output-formats.md\nrendering pipeline"] + CLIVS["cli-vs-api.md\nbatch vs full pipeline"] + TEST["development/testing.md\npytest"] + API["api/index.md\nuvicorn + fastapi install"] + PUB["development/publishing.md\nmkdocs build"] + DEBUG["logging-and-debug.md\nRA_PIPELINE__DEBUG"] + HEUR["llm/heuristic-vs-llm.md\ndefault quality profile"] + end + README --> INSTALL + README --> QUICK + README --> CLI + SETUP --> HEALTH + CLI --> CLIVS + CONTRIB --> PUB +``` + +## Authoritative map + +| Content | Canonical location | Everywhere else | +|---------|-------------------|-----------------| +| Install + first query | [README.md](https://github.com/Ndevu12/Research_Assistant_Model#quick-start) | Link to `#quick-start` / `#usage` anchors | +| Install context (packages, layout) | [getting-started/installation.md](../getting-started/installation.md) | Link to README for bash blocks | +| First-run internals | [getting-started/quick-start.md](../getting-started/quick-start.md) | No repeated usage cheat sheet | +| Setup / health / manager | [setup-system/index.md](../setup-system/index.md) | One-liner + link | +| Health check matrix | [getting-started/health-check.md](../getting-started/health-check.md) | Link to setup-system for commands | +| CLI flags and execution paths | [user-guide/cli.md](../user-guide/cli.md) | Flag table + **unique** examples only; no full README block | +| Output format rendering | [user-guide/output-formats.md](../user-guide/output-formats.md) | Link to README for `--format` commands | +| CLI vs API / provider divergence | [user-guide/cli-vs-api.md](../user-guide/cli-vs-api.md) | Admonition + link; no re-explanation | +| Default heuristic quality | [llm/heuristic-vs-llm.md](../llm/heuristic-vs-llm.md) | Short admonition + link | +| pytest | [development/testing.md](../development/testing.md) | Link only | +| API server | [api/index.md](../api/index.md) | Link only | +| MkDocs / publish / policy check | [development/publishing.md](../development/publishing.md) + root [contributing.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md) | Link only | +| Debug mode | [operations/logging-and-debug.md](../operations/logging-and-debug.md) | Link from troubleshooting | +| Contributor entry | Root [contributing.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md) | [docs/contributing.md](../contributing.md) = pointer only | +| Setup stub | [setups/README.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/setups/README.md) | 3 commands + docs links (keep as-is) | + +## Reference pattern + +Replace duplicated blocks with links. This pattern is already used in [installation.md](../getting-started/installation.md): + +```markdown +## Commands + +Copy-paste steps: [README Quick Start](https://github.com/Ndevu12/Research_Assistant_Model#quick-start). + +## [Topic-specific section — unique to this page] +``` + +For in-site links (MkDocs): + +```markdown +Setup commands: [Setup system](../setup-system/index.md#quick-start). +``` + +Use **stable headings** (`## Quick start`, `## Commands`) as link anchors on canonical pages. + +## What not to duplicate + +- Full README usage blocks (4+ lines of `python -m src`) — [README Usage](https://github.com/Ndevu12/Research_Assistant_Model#usage) only +- Setup health/manager command trio — [setup-system/index.md](../setup-system/index.md) (+ README + setups stub) +- `pipenv run mkdocs serve` — [contributing.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md) and [publishing.md](../development/publishing.md) only +- `pipenv run pytest` blocks — [contributing.md](https://github.com/Ndevu12/Research_Assistant_Model/blob/main/contributing.md) and [testing.md](../development/testing.md) only + +!!! tip "Context-specific one-liners are OK" + Cookbooks and troubleshooting may include a single command when it is recipe- or symptom-specific. Remove only **full duplicate cheat sheets**, not unique examples. + +## CI enforcement + +Pull requests run `scripts/check_docs_policy.py` alongside `mkdocs build --strict`. The script rejects forbidden duplicate command signatures outside their allowed files. See [Publishing documentation](../development/publishing.md) for details. diff --git a/docs/research-quality-known-issues.md b/docs/research-quality-known-issues.md index dbd4685..1753b88 100644 --- a/docs/research-quality-known-issues.md +++ b/docs/research-quality-known-issues.md @@ -1,217 +1,8 @@ # Research Quality — Known Issues -**Status:** P0–P2 fixes **implemented** (2026-05-22) -**Recorded:** 2026-05-22 -**Triggering run:** Interactive query `transformer attention mechanisms` -**Branch context:** `feat/multi-stage-research-pipeline` +!!! note "Moved" + This document lives at **[quality/known-issues.md](quality/known-issues.md)** in the docs site. ---- + Published URL: [Known Issues](https://ndevu12.github.io/Research_Assistant_Model/quality/known-issues/) -## Summary - -Reports could look **factually wrong or off-topic** even when the pipeline completed successfully. For the reference run, this was **not** primarily an Ollama/LLM accuracy failure: **no LLM calls were made** (`llm_tokens_in: 0`, `synthesis.llm_enabled=false`). The pipeline retrieved a mix of relevant and irrelevant papers, ranked some off-topic work highly, and assembled a misleading executive summary using **heuristic synthesis** and **abstract snippet extraction**. - -The fixes below are **query-agnostic** — they use embedding similarity, adaptive corpus-relative thresholds, and generic keyword-collision detection rather than hardcoded NLP/ML domain lists. Multi-domain regression tests live in `tests/test_research_quality.py`. - ---- - -## Reference Run — Observed Symptoms (Pre-Fix) - -| Symptom | Example from output | -|---------|---------------------| -| Executive summary off-topic | Opens with *“Air pollution poses a critical global public health challenge…”* for an NLP/transformer query | -| Irrelevant papers in top results | Cervical cancer (CerviFormer), tea evapotranspiration, Arabic sign language, boring machining (PhyDT) | -| Suspicious canonical paper metadata | *Attention Is All You Need* listed as **2025** with DOI `10.65215/2q58a426` | -| Fragmented clustering | Many themes prefixed `Unclustered: …` (8 thin clusters) | -| Generic analysis placeholders | *“Details inferred from abstract only”*, *“Full disagreement analysis requires LLM synthesis”* | -| No relevance pruning | 25 papers kept, **0 filtered** by relevance stage | -| High retrieval volume | 96 papers retrieved → 76 after dedup → 25 ranked | - -**Pipeline duration:** ~7.4s (11 stages, no failures) -**Debug artifact:** `logs/debug/pipeline_*_20260522_*.json` (query: `transformer attention mechanisms`) -**Ranking top score:** `0.8922` (misleadingly high for topical fit) - ---- - -## Fix Status - -| ID | Component | Status | What changed | -|----|-----------|--------|--------------| -| RC-1 | Query expansion | **Fixed** | Phrase-aware synonyms, Jaccard gate, broad-term guard; ML-specific templates removed | -| RC-2 | Ranking | **Fixed** | Embedding weight 30%; embedding outlier + keyword-collision penalties (no ML branch) | -| RC-3 | Relevance gate | **Fixed** | Adaptive embedding floor (percentile + gap-from-top); configurable `min_embedding_similarity` | -| RC-4 | Synthesis & summary | **Fixed** | Top-quartile agreements; template executive summary wired to config thresholds | -| RC-5 | Clustering | **Fixed** | Macro-cluster merge when HDBSCAN noise > 50%; `Theme:` labels instead of singleton spam | -| RC-6 | Metadata / dedup | **Fixed** | Generic year/DOI sanity; `canonical_boost: 0.0` default (registry opt-in) | -| RC-7 | LLM mode | **Fixed** | `llm_mode: auto \| on \| off` via `resolve_llm_features.py` | - ---- - -## Architecture Context - -``` -Query → expansion (heuristic) → retrieval (OpenAlex + Semantic Scholar) - → dedup → ranking → relevance_scoring → clustering → synthesis (heuristic) - → gap_analysis (heuristic) → report_generation -``` - -| Stage | LLM used? | Quality notes | -|-------|-----------|---------------| -| Query expansion | Auto (off on 3B Ollama) | Heuristic variants with quality gates | -| Retrieval | No | Keyword/API search | -| Ranking | No | Weighted signals + embedding cache | -| Relevance | No | Adaptive embedding floor + concept match | -| Synthesis | Auto | Heuristic path uses embedding-aligned agreements | -| Report | No | Template summary + validated snippet | - ---- - -## Root Cause Analysis (Historical) - -These sections document **why** the reference run failed. Each maps to a fix above. - -### RC-1: Query expansion produces noisy search strings - -**Location:** `src/research/query_expansion.py` - -For query `transformer attention mechanisms`, heuristic expansion previously yielded redundant variants like `attention mechanism attention mechanisms`. Fixes: phrase-aware replacement, Jaccard overlap gate, broad-term guard, concept bigrams. **No** transformer/attention-specific templates. - ---- - -### RC-2: Ranking mis-weighted for topical precision - -**Location:** `src/research/ranking.py`, `config/default.yaml` - -Previously `embedding_similarity` was 5% and `semantic_relevance` was keyword-only. Now embedding weight is 30%, with generic penalties for: - -- Partial core-concept coverage -- Embedding outliers (far below top-5 mean similarity) -- Keyword collision (high overlap, low embedding similarity) - -Config: `outlier_embedding_gap: 0.12`, `keyword_collision_max_sim: 0.40`. - ---- - -### RC-3: Relevance filter was effectively disabled - -**Location:** `src/research/relevance_scoring.py` - -Previously filtered only at `rank_score < 0.05`. Now uses composite gate with adaptive embedding floor: - -```python -floor = max(min_embedding_similarity, percentile(sims, keep_percentile), top_sim - gap_from_top) -``` - -Percentile skipped when corpus size < 8. Default `min_embedding_similarity: 0.35`. - ---- - -### RC-4: Heuristic synthesis built misleading narratives - -**Location:** `src/analysis/synthesis.py`, `src/reporting/report_generation.py` - -Previously `agreements[0]` from rank order drove the executive summary. Now: - -- Agreements drawn from top embedding-similarity quartile -- Executive summary uses query + cluster themes template -- Validated snippet only when embedding similarity ≥ configured floor - ---- - -### RC-5: Clustering fragmented into “Unclustered” singletons - -**Location:** `src/research/clustering.py` - -When HDBSCAN noise ratio > `noise_merge_threshold` (0.5), noise papers merge into ≤4 keyword macro-clusters labeled `Theme: …`. - ---- - -### RC-6: Scholarly metadata quality - -**Location:** `src/research/metadata_sanity.py`, `src/retrieval/deduplication.py` - -Generic rules: future-year correction, DOI format checks, richer duplicate preference. Canonical work registry is **opt-in** (`canonical_boost: 0.0` default). - ---- - -### RC-7: LLM synthesis disabled by default - -**Location:** `src/config/resolve_llm_features.py`, `config/default.yaml` - -Tri-state `llm_mode: auto` enables LLM on cloud providers and capable local models (8B+). Small Ollama models (e.g. `llama3.2:3b`) stay heuristic-only unless explicitly overridden. - ---- - -## Quality Modes - -| Mode | Typical setup | Synthesis | Expansion | -|------|---------------|-----------|-----------| -| Heuristic-only | `llama3.2:3b`, `llm_mode: auto` | Off | Off | -| Balanced local | `llama3.1:8b`, `llm_mode: auto` | On | On | -| Cloud | `openai` / `anthropic`, `llm_mode: auto` | On | On | - -Override with `RA_SYNTHESIS__LLM_MODE=on|off|auto` or legacy `RA_SYNTHESIS__LLM_ENABLED=true|false`. - ---- - -## Reproduction Checklist - -Run multi-domain regression (no network): - -```bash -pipenv run pytest tests/test_research_quality.py -q -``` - -Full pipeline smoke test: - -```bash -pipenv run python -m src "transformer attention mechanisms" -``` - -**Expect (post-fix):** - -- [ ] Executive summary uses query + cluster themes; no off-topic decoy domain in lead -- [ ] Keyword-collision decoys demoted in ranking and filtered by relevance -- [ ] High HDBSCAN noise yields ≤4 `Theme:` macro-clusters, not N singletons -- [ ] `llm_tokens_in: 0` when running heuristic-only mode on 3B model - -**Inspect expansion:** - -```bash -pipenv run python -c " -from src.research.query_expansion import expand_query_heuristic, extract_core_concepts -q = 'transformer attention mechanisms' -print(expand_query_heuristic(q, extract_core_concepts(q))) -" -``` - ---- - -## Related Configuration - -| Setting | Default | Quality impact | -|---------|---------|----------------| -| `synthesis.llm_mode` | `auto` | Model-aware LLM enable (RC-7) | -| `query_expansion.llm_mode` | `auto` | Model-aware expansion (RC-1) | -| `ranking.weights.embedding_similarity` | `0.30` | Primary semantic steering (RC-2) | -| `ranking.outlier_embedding_gap` | `0.12` | Demote embedding outliers (RC-2) | -| `ranking.keyword_collision_max_sim` | `0.40` | Keyword-without-semantics penalty (RC-2) | -| `ranking.canonical_boost` | `0.0` | Opt-in landmark boost (RC-6) | -| `relevance_scoring.min_embedding_similarity` | `0.35` | Base embedding floor (RC-3) | -| `relevance_scoring.adaptive_embedding` | `true` | Corpus-relative floor (RC-3) | -| `clustering.noise_merge_threshold` | `0.5` | Macro-cluster merge trigger (RC-5) | - -Environment overrides: - -- `RA_RELEVANCE_SCORING__MIN_EMBEDDING_SIMILARITY` -- `RA_SYNTHESIS__LLM_MODE=auto` -- `RA_RANKING__CANONICAL_BOOST=0.05` (opt-in) - ---- - -## Change Log - -| Date | Action | -|------|--------| -| 2026-05-22 | Initial documentation from interactive run analysis | -| 2026-05-22 | P0–P2 fixes implemented; multi-domain tests in `test_research_quality.py` | +Please edit `docs/quality/known-issues.md` — this file is kept only as a redirect for existing links. diff --git a/docs/retrieval/overview.md b/docs/retrieval/overview.md new file mode 100644 index 0000000..0c4c747 --- /dev/null +++ b/docs/retrieval/overview.md @@ -0,0 +1,103 @@ +# Retrieval Overview + +The retrieval layer searches multiple scholarly APIs in parallel, normalizes responses into `RetrievedPaper` objects, and feeds the embedding-backed pipeline stages. + +Source: `src/retrieval/providers/`, `src/retrieval/retrieval_stage.py`, `src/retrieval/orchestrator.py`. + +## Architecture + +```mermaid +flowchart TD + QE[Query expansion variants] --> RS[RetrievalStage] + RS --> REG[get_enabled_providers] + REG --> OA[OpenAlex] + REG --> SS[Semantic Scholar] + REG --> AX[arXiv / CrossRef / …] + OA --> MERGE[Merge + normalize] + SS --> MERGE + AX --> MERGE + MERGE --> DEDUP[Deduplication stage] +``` + +1. **Registry** — `get_enabled_providers(settings)` iterates `settings.retrieval.providers`, skips disabled names, instantiates classes from `_PROVIDER_CLASSES` (`registry.py`). +2. **Concurrency** — For each expanded query variant, all enabled providers run in parallel via `asyncio.gather`. A semaphore caps total concurrent searches at `retrieval.concurrency_limit` (default 4). +3. **Limits** — Each provider receives `settings.retrieval.per_provider_limit` (default 8). Per-provider `limit` in YAML is **not** used. +4. **Failure handling** — `_safe_search()` catches exceptions per provider → warning log + empty list. The stage is marked **partial** if any provider fails. +5. **Cache bypass** — When `memory.cache_enabled=true` and a cache hit exists, `cached_papers` is passed as an initial artifact and network retrieval is skipped. + +## CLI vs full pipeline + +!!! info "CLI vs full pipeline" + `python -m src "query"` calls `run_research_helper()`, which **hardcodes** OpenAlex + Semantic Scholar only. YAML provider toggles are ignored for the provider set. + +| Aspect | Full pipeline (`run_research` / API) | CLI batch (`run_research_helper`) | +|--------|--------------------------------------|-----------------------------------| +| Entry | API, interactive session, programmatic | `python -m src "query"` | +| Settings | Full `AppSettings()` merge | Constructor override replaces `providers` dict | +| Providers | All `enabled=true` in config | OpenAlex + Semantic Scholar only | +| Limit | `per_provider_limit` (default 8) | `k_each` param (default 8) | +| arXiv / CrossRef | Honored when enabled in YAML/env | **Not available** | + +To use additional providers from the CLI path today, use **interactive mode**, the **API**, or call `run_research()` programmatically. See [CLI vs API](../user-guide/cli-vs-api.md) and [Provider matrix](provider-matrix.md). + +```python +# orchestrator.py — CLI helper override +settings = AppSettings( + retrieval={ + "per_provider_limit": k_each, + "providers": { + "openalex": {"enabled": True, "limit": k_each}, + "semantic_scholar": {"enabled": True, "limit": k_each}, + }, + } +) +``` + +Interactive mode (`python -m src` with no query arg) uses the **full pipeline** via `InteractiveResearchSession.run_full_query()` → `run_research_with_result()`. + +## HTTP client + +- Library: **aiohttp** (`ClientSession` created in `RetrievalStage`) +- Search timeout: **60 seconds** per request +- Health check timeout: **15 seconds** +- Retries: **3** attempts with exponential backoff (`2 ** attempt` seconds) +- Rate limits: Semantic Scholar and CrossRef honor HTTP **429** + `Retry-After` header + +## Enabling providers + +**YAML** (`config/providers.yaml`): + +```yaml +providers: + arxiv: + enabled: true + crossref: + enabled: true +``` + +**Environment:** + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +RA_CROSSREF_MAILTO=you@example.com +S2_API_KEY=optional_key_for_higher_limits +``` + +**Programmatic:** + +```python +from src.config.settings import AppSettings +from src.retrieval.orchestrator import run_research + +settings = AppSettings( + retrieval={"providers": {"arxiv": {"enabled": True}}} +) +report = await run_research("transformer attention", settings=settings) +``` + +## Extensibility + +Custom providers register via `register_provider(name, class)` in `registry.py`. See [Extensibility](../development/extensibility.md) and `tests/test_phase3_extensibility.py`. + +See also: [Provider matrix](provider-matrix.md), [Configuration — retrieval](../configuration/yaml-reference.md#retrieval), [Retrieval stage](../architecture/stages/retrieval.md). diff --git a/docs/retrieval/per-provider/arxiv.md b/docs/retrieval/per-provider/arxiv.md new file mode 100644 index 0000000..2d0cb3b --- /dev/null +++ b/docs/retrieval/per-provider/arxiv.md @@ -0,0 +1,75 @@ +# arXiv + +arXiv covers preprints and e-prints, especially strong for CS, ML, and physics. Disabled by default — enable for the **full pipeline** or API (not the CLI batch shortcut). + +Implementation: `src/retrieval/providers/arxiv.py` + +## HTTP API + +| Attribute | Value | +|-----------|-------| +| Search URL | `GET https://export.arxiv.org/api/query` | +| Query params | `search_query`, `start=0`, `max_results` | +| Response format | Atom XML | +| Authentication | None | +| Timeout | 60s (search), 15s (health) | +| Retries | 3 with exponential backoff | + +### Query syntax + +The provider preserves arXiv field syntax when present: + +| User query | `search_query` sent | +|------------|---------------------| +| `ti:attention abs:transformer` | unchanged | +| `transformer attention` | `all:transformer attention` | + +Field prefixes: `ti:` (title), `abs:` (abstract), `au:` (author), etc. + +### Example request + +```http +GET https://export.arxiv.org/api/query?search_query=all:transformer+attention&start=0&max_results=8 +``` + +## Normalization + +Atom entries parse to `RetrievedPaper`: + +| Atom element | `RetrievedPaper` field | +|--------------|------------------------| +| `title` | `title` | +| `summary` | `abstract` | +| `published` year | `year` | +| `arxiv:primary_category` | `venue` (category label) | +| PDF link or `id` | `url` | +| `arxiv:doi` if present | `doi` | +| Author list | `authors` | + +## Configuration + +```yaml +retrieval: + providers: + arxiv: + enabled: true +``` + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +``` + +!!! info "CLI limitation" + `python -m src "query"` does not enable arXiv. Use interactive mode, the API, or `run_research()` with custom settings. + +## Health check + +Search with `max_results=1`, 15s timeout. + +## Operational notes + +- arXiv requests a 3-second delay between calls in their terms of use — the pipeline's concurrency limit helps avoid hammering the API +- Preprint-heavy results may rank differently than peer-reviewed venues from OpenAlex/CrossRef +- Good complement when researching very recent ML work not yet indexed elsewhere + +See also: [Provider matrix](../provider-matrix.md), [Configuration cookbook](../../user-guide/configuration-cookbook.md). diff --git a/docs/retrieval/per-provider/crossref.md b/docs/retrieval/per-provider/crossref.md new file mode 100644 index 0000000..d87c498 --- /dev/null +++ b/docs/retrieval/per-provider/crossref.md @@ -0,0 +1,77 @@ +# CrossRef + +CrossRef indexes DOI-registered scholarly works across publishers. Disabled by default — enable for full pipeline or API runs. + +Implementation: `src/retrieval/providers/crossref.py` + +## HTTP API + +| Attribute | Value | +|-----------|-------| +| Search URL | `GET https://api.crossref.org/works` | +| Query params | `query`, `rows` | +| Authentication | Polite pool via `User-Agent` mailto (recommended) | +| Timeout | 60s (search), 15s (health) | +| Retries | 3 with exponential backoff | +| Rate limiting | HTTP 429 → sleep `Retry-After` | + +### User-Agent / polite pool + +CrossRef recommends identifying your client with a contact email: + +| Variable | Header value | +|----------|--------------| +| `RA_CROSSREF_MAILTO` | `ResearchAssistant/1.0 (mailto:you@example.com)` | +| `CROSSREF_MAILTO` | Alias for the above | + +Without mailto, requests use `ResearchAssistant/1.0` — functional but lower polite-pool priority. + +### Example request + +```http +GET https://api.crossref.org/works?query=transformer+attention&rows=8 +User-Agent: ResearchAssistant/1.0 (mailto:you@example.com) +``` + +## Normalization + +CrossRef work items map to `RetrievedPaper`: + +| CrossRef field | `RetrievedPaper` field | +|----------------|------------------------| +| `title[]` | `title` (first element) | +| `abstract` | `abstract` (HTML stripped) | +| `published-print` / `published-online` / `issued` | `year` | +| `container-title[]` | `venue` | +| `URL` or DOI link | `url` | +| `DOI` | `doi` | +| `author[]` | `authors` (given + family) | + +## Configuration + +```yaml +retrieval: + providers: + crossref: + enabled: true +``` + +```bash +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +RA_CROSSREF_MAILTO=you@example.com +``` + +!!! tip "Always set mailto" + CrossRef polite pool improves reliability under load. Set `RA_CROSSREF_MAILTO` before enabling this provider in production or batch jobs. + +## Health check + +Query with `rows=1`, 15s timeout. + +## Operational notes + +- Strong for DOI resolution and publisher metadata; abstracts may be missing for some records +- Pairs well with OpenAlex (coverage) and arXiv (recent preprints) +- 429 handling matches Semantic Scholar — automatic backoff + +See also: [Provider matrix](../provider-matrix.md), [Environment variables](../../configuration/environment-variables.md). diff --git a/docs/retrieval/per-provider/openalex.md b/docs/retrieval/per-provider/openalex.md new file mode 100644 index 0000000..7d14837 --- /dev/null +++ b/docs/retrieval/per-provider/openalex.md @@ -0,0 +1,67 @@ +# OpenAlex + +OpenAlex is the primary bibliographic index provider — enabled by default and used in both the CLI shortcut path and the full pipeline. + +Implementation: `src/retrieval/providers/openalex.py` + +## HTTP API + +| Attribute | Value | +|-----------|-------| +| Search URL | `GET https://api.openalex.org/works` | +| Query params | `search=`, `per-page=` | +| Authentication | None required | +| Timeout | 60s (search), 15s (health) | +| Retries | 3 with exponential backoff | + +### Example request + +```http +GET https://api.openalex.org/works?search=transformer+attention&per-page=8 +``` + +No API key or mailto is required. OpenAlex is suitable for broad scholarly coverage including DOIs, venues, and citation counts. + +## Normalization + +OpenAlex work JSON maps to `RetrievedPaper`: + +| OpenAlex field | `RetrievedPaper` field | +|----------------|------------------------| +| `display_name` | `title` | +| `abstract_inverted_index` | `abstract` (reconstructed) | +| `publication_year` | `year` | +| `primary_location.source.display_name` | `venue` | +| `id` or landing page URL | `url` | +| `doi` | `doi` | +| `cited_by_count` | `citation_count` | + +Full raw JSON is preserved in `raw_metadata`. + +## Configuration + +**Default:** enabled in `config/default.yaml` and `config/providers.yaml`. + +```yaml +retrieval: + providers: + openalex: + enabled: true +``` + +```bash +RA_RETRIEVAL__PROVIDERS__OPENALEX__ENABLED=true +RA_RETRIEVAL__PER_PROVIDER_LIMIT=10 +``` + +## Health check + +`_ping()` issues a minimal `per-page=1` query against the works endpoint. Used by setup health checks when retrieval providers are validated. + +## Operational notes + +- No explicit 429 handler — relies on retry + backoff +- Rate limits are generous for polite use; avoid tight loops in batch scripts +- OpenAlex IDs (`https://openalex.org/W...`) are used as paper URLs when no landing page exists + +See also: [Provider matrix](../provider-matrix.md), [Retrieval overview](../overview.md). diff --git a/docs/retrieval/per-provider/semantic-scholar.md b/docs/retrieval/per-provider/semantic-scholar.md new file mode 100644 index 0000000..1130ad4 --- /dev/null +++ b/docs/retrieval/per-provider/semantic-scholar.md @@ -0,0 +1,70 @@ +# Semantic Scholar + +Semantic Scholar provides paper metadata and abstracts via the bulk search API. Enabled by default alongside OpenAlex. + +Implementation: `src/retrieval/providers/semantic_scholar.py` + +## HTTP API + +| Attribute | Value | +|-----------|-------| +| Search URL | `GET https://api.semanticscholar.org/graph/v1/paper/search/bulk` | +| Query params | `query`, `limit`, `fields=title,abstract,year,venue,url,externalIds` | +| Authentication | Optional — `S2_API_KEY` env → `x-api-key` header | +| Timeout | 60s (search), 15s (health) | +| Retries | 3 with exponential backoff | +| Rate limiting | HTTP 429 → sleep `Retry-After` (default 60s) | + +### Example request + +```http +GET https://api.semanticscholar.org/graph/v1/paper/search/bulk?query=transformer+attention&limit=8&fields=title,abstract,year,venue,url,externalIds +x-api-key: YOUR_KEY # optional +``` + +### API key + +| Variable | Required | Effect | +|----------|----------|--------| +| `S2_API_KEY` | No | Higher rate limits when set | + +Without a key, anonymous rate limits apply. For heavy interactive use or batch jobs, set `S2_API_KEY` in `.env`. + +## Normalization + +| S2 field | `RetrievedPaper` field | +|----------|------------------------| +| `title` | `title` | +| `abstract` | `abstract` | +| `year` | `year` | +| `venue` | `venue` | +| `url` | `url` | +| `externalIds.DOI` | `doi` (prefixed with `https://doi.org/` if needed) | + +Citation count is not always present in the requested field set; ranking may rely more on embedding similarity for S2-sourced papers. + +## Configuration + +```yaml +retrieval: + providers: + semantic_scholar: + enabled: true +``` + +```bash +S2_API_KEY=your_api_key_here +RA_RETRIEVAL__PROVIDERS__SEMANTIC_SCHOLAR__ENABLED=true +``` + +## Health check + +Minimal search with `limit=1` against the bulk endpoint, 15s timeout. + +## Operational notes + +- Watch for 429 responses in logs — the client waits and retries automatically +- Bulk search is used (not single-paper lookup) for query-variant parallelism +- Combined with OpenAlex, provides good recall for CS/ML topics + +See also: [Provider matrix](../provider-matrix.md), [Environment variables](../../configuration/environment-variables.md#retrieval-ra_retrieval__). diff --git a/docs/retrieval/provider-matrix.md b/docs/retrieval/provider-matrix.md new file mode 100644 index 0000000..6577d85 --- /dev/null +++ b/docs/retrieval/provider-matrix.md @@ -0,0 +1,93 @@ +# Provider Matrix + +Summary of all seven registered retrieval providers. HTTP details are in per-provider pages. + +Source: `src/retrieval/providers/registry.py`, `_PROVIDER_CLASSES`. + +## Status overview + +| Key | Status | Default enabled | Auth | Notes | +|-----|--------|-----------------|------|-------| +| [openalex](per-provider/openalex.md) | **Live** | Yes | None | Used in CLI and full pipeline | +| [semantic_scholar](per-provider/semantic-scholar.md) | **Live** | Yes | `S2_API_KEY` (optional) | Used in CLI and full pipeline | +| [arxiv](per-provider/arxiv.md) | **Live** | No | None | Full pipeline / API only | +| [crossref](per-provider/crossref.md) | **Live** | No | Mailto in User-Agent (recommended) | Full pipeline / API only | +| `pubmed` | **Stub** | No | — | `NotImplementedError` if enabled | +| `core` | **Stub** | No | — | `NotImplementedError` if enabled | +| `dblp` | **Stub** | No | — | `NotImplementedError` if enabled | + +!!! warning "Stub providers" + PubMed, CORE, and DBLP are registered but not implemented. Enabling them raises `NotImplementedError` in `search()`, which `_safe_search()` catches — you get a warning and empty results for that provider, not a crash. + +## Enable snippets + +**OpenAlex + Semantic Scholar (defaults — no change needed):** + +```yaml +# config/providers.yaml +providers: + openalex: + enabled: true + semantic_scholar: + enabled: true +``` + +**Add arXiv and CrossRef:** + +```yaml +providers: + arxiv: + enabled: true + crossref: + enabled: true +``` + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +RA_CROSSREF_MAILTO=you@example.com +``` + +**Do not enable stubs:** + +```yaml +# Will warn and return no papers from this provider +pubmed: + enabled: true # not recommended until implemented +``` + +## HTTP summary + +| Provider | Method | Base endpoint | Search timeout | 429 handling | +|----------|--------|---------------|----------------|--------------| +| OpenAlex | GET | `https://api.openalex.org/works` | 60s | Retry only | +| Semantic Scholar | GET | `https://api.semanticscholar.org/graph/v1/paper/search/bulk` | 60s | Sleep `Retry-After` | +| arXiv | GET | `https://export.arxiv.org/api/query` | 60s | Retry only | +| CrossRef | GET | `https://api.crossref.org/works` | 60s | Sleep `Retry-After` | + +All live providers: **3 retries**, exponential backoff, health ping at **15s** timeout. + +## Config keys used at runtime + +| Key | Default | Used? | +|-----|---------|-------| +| `retrieval.concurrency_limit` | `4` | Yes — caps parallel variant searches | +| `retrieval.per_provider_limit` | `8` | Yes — results per provider per variant | +| `retrieval.providers..enabled` | see table | Yes (full pipeline only) | +| `retrieval.providers..limit` | `8` | **No** — overridden by `per_provider_limit` | + +## Graceful degradation + +```mermaid +flowchart TD + Q[Query variant] --> GATHER[asyncio.gather enabled providers] + GATHER --> P1[Provider search] + P1 -->|success| MERGE[Merge papers] + P1 -->|exception| W[Warning + empty list] + W --> MERGE + MERGE --> DEDUP[Deduplication] +``` + +Partial retrieval (some providers failed) still proceeds through the pipeline. Check stderr warnings or enable debug dumps to inspect per-provider failures. + +See also: [Retrieval overview](overview.md), [Environment variables](../configuration/environment-variables.md#retrieval-ra_retrieval__). diff --git a/docs/setup-system/index.md b/docs/setup-system/index.md new file mode 100644 index 0000000..253d578 --- /dev/null +++ b/docs/setup-system/index.md @@ -0,0 +1,207 @@ +# Setup System + + + +Automated setup for local Ollama inference: dependency checks, Ollama install, model pull, and health reporting. Cloud LLM providers (`openai`, `anthropic`) skip this path entirely. + +Source: `setups/` package, invoked from `ensure_setup()` in `src/__main__.py`. + +## Quick start + +### Check setup status + +```bash +pipenv run python -m setups.health_check +``` + +### Run full setup + +```bash +pipenv run python -m setups.manager +``` + +### Run setup with a specific model + +```bash +pipenv run python -m setups.manager --model llama3.1:8b +``` + +## How CLI startup uses setup + +Every `python -m src` run calls `ensure_setup()` before the pipeline: + +```mermaid +flowchart TD + CLI["python -m src"] --> PIP{PIPENV_ACTIVE?} + PIP -->|no| WARN[Log Pipenv warning] + PIP --> PROV{llm.provider == ollama?} + WARN --> PROV + PROV -->|no| OK[Return True — skip setup] + PROV -->|yes| HC[health_check.check_ollama_running + check_model_available] + HC --> R{both OK?} + R -->|yes| OK + R -->|no| REP[health_check.print_report] + REP --> MGR[manager.run_setup] + MGR --> S{success?} + S -->|yes| OK + S -->|no| EXIT[sys.exit 1] +``` + +Implementation notes from `src/__main__.py`: + +1. Adds `setups/` to `sys.path` and imports `health_check` + `manager` dynamically +2. Skips all Ollama checks when `RA_LLM__PROVIDER` is not `ollama` +3. On failure, runs `manager.run_setup()` automatically (first-run experience) +4. Exits with code `1` if auto-setup fails — user must run manager manually + +See [Health check](../getting-started/health-check.md) for the check matrix. + +## Manager orchestration + +`manager.run_setup(model_name=None)` runs three steps in order (`setups/manager.py`): + +| Step | Function | Purpose | +|------|----------|---------| +| Python Dependencies | `setup.install_python_deps()` | Ensures Pipenv deps (aiohttp, etc.) | +| Ollama Installation | `ollama.install_ollama()` | Downloads/installs Ollama binary if missing | +| Ollama Model Setup | `ollama.setup_ollama(model_name)` | Pulls and validates the target model | + +Returns `True` only when all steps succeed. Logs a summary with the configured model name. + +```bash +pipenv run python -m setups.manager +pipenv run python -m setups.manager --model mistral # must be in ollama_models.yaml +``` + +## Module reference + +### `health_check.py` + +Validates runtime readiness. Used by CLI startup and manual diagnostics. + +| Function | Checks | +|----------|--------| +| `check_python_deps()` | `pipenv` on PATH, `aiohttp` importable | +| `check_embedding_deps()` | `sentence-transformers` importable | +| `check_ollama_installed()` | `ollama` binary on PATH | +| `check_ollama_running()` | `ollama list` succeeds (5s timeout) | +| `check_model_available(model)` | Resolved model exists locally | + +Model resolution uses `resolve_target_model()` from `src/config/model_selection.py`: + +- Reads `RA_LLM__MODEL` (default `auto`) +- When `auto`, selects best fit from `config/ollama_models.yaml` based on RAM/disk +- Validates via `ollama list` and `is_model_installed()` + +```bash +pipenv run python -m setups.health_check +pipenv run python -m setups.health_check --model llama3.1:8b +``` + +`print_report()` prints all checks and returns overall pass/fail. + +### `setup.py` + +Installs Python dependencies via Pipenv: + +```bash +pipenv run python -m setups.setup +``` + +### `ollama.py` + +Ollama binary and model management: + +```bash +pipenv run python -m setups.ollama install +pipenv run python -m setups.ollama setup +pipenv run python -m setups.ollama setup --model llama3.2:3b +``` + +### `manager.py` + +Orchestrates all setup steps. Entry point for full setup and CLI auto-recovery. + +## Python API + +```python +from setups import run_setup, print_report, check_ollama_running + +is_running, message = check_ollama_running() +all_ok = print_report() +success = run_setup(model_name="llama3.2:3b") +``` + +Re-exports from `setups/__init__.py` for programmatic use in tests or custom tooling. + +## Supported models + +Catalog: `config/ollama_models.yaml`. Each entry defines RAM/disk requirements, priority, and optional feature flags (e.g. enabling LLM synthesis on 8B models). + +```bash +# Auto-select (recommended) +pipenv run python -m setups.manager + +# Force a supported model +pipenv run python -m setups.manager --model llama3.1:8b + +# Or in .env (takes precedence over YAML defaults) +# RA_LLM__MODEL=llama3.1:8b +``` + +Auto-selection picks the highest-priority model that fits available RAM and disk. See [Ollama](../llm/ollama.md) for catalog details and `llm_mode` interaction. + +## Logging + +Setup operations log to `logs/`: + +| File | Contents | +|------|----------| +| `combined_YYYYMMDD.log` | All log levels | +| `error_YYYYMMDD.log` | Errors and warnings | +| `events_YYYYMMDD.log` | Structured events | + +```bash +tail -f logs/combined_*.log +``` + +Uses `src/utils/logging_system.py` — same logger as the research pipeline. + +## Troubleshooting + +### Setup fails with "pipenv not found" + +```bash +pip install pipenv +cd /path/to/Research_Assistant_Model +pipenv install +``` + +### Setup fails with "Ollama not found" + +Install from [ollama.com/download](https://ollama.com/download), then: + +```bash +pipenv run python -m setups.manager +``` + +### Model pull fails + +Check network and retry: + +```bash +pipenv run python -m setups.ollama setup --model llama3.2:3b +``` + +### Embedding deps missing + +```bash +pipenv install # installs sentence-transformers from Pipfile +pipenv run python -m setups.health_check +``` + +### Cloud provider — setup skipped + +When `RA_LLM__PROVIDER=openai` or `anthropic`, `ensure_setup()` returns immediately. Validate API keys separately — no Ollama required. + +See also: [Health check](../getting-started/health-check.md), [Troubleshooting](../operations/troubleshooting.md), [Installation](../getting-started/installation.md). diff --git a/docs/user-guide/cli-vs-api.md b/docs/user-guide/cli-vs-api.md new file mode 100644 index 0000000..9ec6ee1 --- /dev/null +++ b/docs/user-guide/cli-vs-api.md @@ -0,0 +1,181 @@ +# CLI vs API — Execution Paths + +The same research pipeline can be reached through four entry points. They differ in **which settings apply**, **which retrieval providers run**, and **how output is delivered**. This page is the authoritative comparison — use it when batch CLI behavior does not match your YAML config. + +## Entry point map + +```mermaid +flowchart TD + subgraph cli [CLI src/__main__.py] + B["Batch: python -m src \"query\""] + I["Interactive: python -m src"] + end + subgraph other [Programmatic / HTTP] + P["run_research() / run_research_with_result()"] + A["POST /research FastAPI"] + end + B --> H[run_research_helper] + I --> S[InteractiveResearchSession] + S --> W[run_research_with_result] + P --> W + A --> BP[build_pipeline] + BP --> W + H --> W2[run_research_with_result] + W --> PL[ResearchPipeline 11 stages] + W2 --> PL +``` + +| Entry | Source | Settings object | Provider set | +|-------|--------|-----------------|--------------| +| CLI batch | `run_research_helper()` | **Constructor override** — replaces `retrieval.providers` | OpenAlex + Semantic Scholar only | +| CLI interactive | `InteractiveResearchSession.run_full_query()` | Full `AppSettings()` merge | All `enabled: true` in config | +| Programmatic | `run_research()` / `run_research_with_result()` | Caller-supplied or default `AppSettings()` | All enabled in config | +| HTTP API | `build_pipeline(resolved_settings)` | `get_settings()` at app startup | All enabled in config | + +!!! warning "Batch CLI ignores provider YAML" + `run_research_helper()` builds a fresh `AppSettings` with only OpenAlex and Semantic Scholar. Env vars like `RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true` have **no effect** on `python -m src "query"`. + +## Code trace — batch shortcut + +Batch mode is the default when a positional query is provided: + +```python +# src/__main__.py — batch path +asyncio.run( + run_research_helper( + args.query, + output_format=output_format, + export_formats=export_formats, + output_path=getattr(args, "output", None), + stream_progress=stream_progress, + ) +) +``` + +Inside the helper, settings are **hardcoded**: + +```python +# src/retrieval/orchestrator.py — run_research_helper() +settings = AppSettings( + retrieval={ + "per_provider_limit": k_each, + "providers": { + "openalex": {"enabled": True, "limit": k_each}, + "semantic_scholar": {"enabled": True, "limit": k_each}, + }, + } +) +report, result = await run_research_with_result(user_text, settings=settings, ...) +``` + +Everything else in `AppSettings` (LLM provider, synthesis mode, stage toggles, ranking weights) still comes from YAML/env — only the **provider dict** is replaced. + +## Code trace — full pipeline + +Interactive mode, the API, and programmatic calls use the merged settings: + +```python +# InteractiveResearchSession.run_full_query() +report, pipeline_result = await run_research_with_result( + query, + settings=self.settings, # AppSettings() — full merge + session=self.session, + store=self.store, +) + +# FastAPI POST /research +pipeline = build_pipeline(resolved_settings) +result = await pipeline.execute(request.query) +``` + +`run_research_with_result()` always calls `build_pipeline(resolved_settings)` with whatever settings object was passed. Retrieval uses `get_enabled_providers(settings)` from the registry. + +## Side-by-side comparison + +| Aspect | CLI batch | CLI interactive | API | Programmatic | +|--------|-----------|-----------------|-----|--------------| +| Command / call | `python -m src "q"` | `python -m src` | `POST /research` | `await run_research(...)` | +| Pipeline stages | All 11 (if enabled in config) | All 11 | All 11 | All 11 | +| Retrieval providers | OA + S2 only | Config-enabled | Config-enabled | Config-enabled | +| `per_provider_limit` | `k_each` (default 8) | From config | From config | From config | +| Output delivery | stdout / `--output` | stdout per query | JSON response | Return value | +| Formats | markdown, json, html, pdf | Same | markdown, json, html only | Caller renders | +| Citation `--export` | After report on stdout | Startup flags + follow-up `export …` | `export` request field | Via `render_report_output` | +| Session memory | Not wired (see below) | SQLite + follow-ups | Not exposed | Optional `session`/`store` args | +| Progress | stderr (unless `--no-progress`) | stderr | Not streamed | Configurable | +| Setup gate | `ensure_setup()` always | Same | Not automatic | Caller responsibility | + +## When to use which path + +| Goal | Recommended path | +|------|------------------| +| Quick one-off query, default providers OK | CLI batch | +| Enable arXiv, CrossRef, or stub providers | Interactive, API, or programmatic | +| Follow-up filters without re-retrieval | CLI interactive | +| Integrate with another service | FastAPI or `run_research()` | +| Script JSON output | CLI batch `--format json` (OA+S2) or API | +| Validate provider config changes | Interactive or API — not batch CLI | + +See [Configuration cookbook — Enable arXiv + CrossRef](configuration-cookbook.md#4-enable-arxiv-crossref-full-pipeline). + +## `--session` flag (batch) + +The CLI defines `--session` to enable SQLite memory in batch mode: + +```bash +pipenv run python -m src --session "your query" +``` + +!!! warning "Not yet wired in __main__.py" + As of the current codebase, `args.session` is parsed but **not passed** to `run_research_helper()`. Batch runs do not persist to SQLite despite the flag. Use **interactive mode** for session memory and follow-ups, or call `run_research_with_result(..., session=..., store=...)` programmatically. + +## Output and rendering + +All paths converge on `render_report_output()` in `src/reporting/output.py` for format-specific rendering. Differences: + +- **CLI batch** — prints to stdout; writes `--output` file; appends `--export` blocks after the main report +- **Interactive** — prints each result; follow-ups re-render from cached `last_report` +- **API** — returns both `report` (JSON dict) and `rendered` (string or nested JSON) in the response body + +See [Output formats](output-formats.md) for renderer details. + +## Configuration that applies everywhere + +These settings apply on **all** paths (including batch CLI), because they are not overridden by `run_research_helper()`: + +- LLM provider and model (`RA_LLM__*`) +- Synthesis / query expansion mode (`RA_SYNTHESIS__*`, `RA_QUERY_EXPANSION__*`) +- Pipeline stage toggles (`RA_PIPELINE__ENABLED_STAGES__*`) +- Ranking and relevance weights +- Debug and progress flags (`RA_PIPELINE__DEBUG`, `RA_PIPELINE__STREAM_PROGRESS`) + +Only **retrieval provider selection** differs on the batch shortcut. + +## Programmatic example — full provider control + +```python +import asyncio +from src.config.settings import AppSettings +from src.retrieval.orchestrator import run_research_with_result +from src.reporting.output import render_report_output + +async def main(): + settings = AppSettings() # loads YAML + env + settings.retrieval.providers["arxiv"].enabled = True + + report, result = await run_research_with_result( + "transformer attention mechanisms", + settings=settings, + ) + print(render_report_output(report, output_format="markdown", partial=result.partial)) + +asyncio.run(main()) +``` + +## Related pages + +- [CLI reference](cli.md) — flags and startup behavior +- [Interactive sessions](interactive-sessions.md) — follow-up commands and SQLite +- [API overview](../api/index.md) — install and response shape +- [Retrieval overview](../retrieval/overview.md) — provider registry and concurrency +- [Configuration cookbook](configuration-cookbook.md) — recipes per use case diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md new file mode 100644 index 0000000..6e42690 --- /dev/null +++ b/docs/user-guide/cli.md @@ -0,0 +1,122 @@ +# CLI Reference + + + +## Commands + +This page is the **canonical CLI reference** for flags, execution paths, and examples. + +Quick copy-paste invocations: [README Usage](https://github.com/Ndevu12/Research_Assistant_Model#usage). + +Entry point: `src/__main__.py` → `main()`. + +## Arguments and flags + +| Argument / flag | Description | +|-----------------|-------------| +| `query` | Optional positional research query. Omitted → interactive mode. | +| `--format` | Output format: `markdown` (default), `json`, `html`, `pdf` | +| `--export FORMATS` | Comma-separated citations: `bibtex`, `apa`, `mla`, `chicago` | +| `--output`, `-o` | Write report to file (recommended for html/pdf) | +| `--session` | Parsed but **not wired** — use interactive mode for SQLite memory (see [CLI vs API](cli-vs-api.md)) | +| `--no-progress` | Disable live stderr progress | +| `--help` | Show argparse help | + +### Examples + +```bash +# Markdown to stdout (default) +pipenv run python -m src "graph neural networks survey" + +# JSON structured report +pipenv run python -m src --format json "graph neural networks survey" + +# Print-ready PDF HTML +pipenv run python -m src --format pdf -o reports/out.pdf.html "query" + +# Citations appended after report +pipenv run python -m src --export bibtex,apa,chicago "query" + +# Quiet run (CI, scripting) +pipenv run python -m src --no-progress "query" +``` + +## Execution paths + +### Batch mode (`query` provided) + +``` +main() + → ensure_setup() # Ollama setup if needed + → run_research_helper() # CLI shortcut path + → render_report_output() + → print (or write --output file) +``` + +!!! info "CLI vs full pipeline" + Batch mode hardcodes **OpenAlex + Semantic Scholar** in `run_research_helper()`. Provider YAML/env toggles do not add arXiv or CrossRef on this path. Interactive mode uses the full settings merge. + + Full comparison: [CLI vs API](cli-vs-api.md). + +### Interactive mode (no `query`) + +``` +main() + → ensure_setup() + → run_interactive_mode() + → InteractiveResearchSession + → run_full_query() / handle_follow_up() +``` + +Uses `run_research_with_result()` with full `AppSettings` — honors all enabled retrieval providers. + +See [Interactive sessions](interactive-sessions.md). + +## Startup behavior + +### Pipenv warning + +If `PIPENV_ACTIVE != 1`, a warning is logged recommending `pipenv run python -m src`. + +### Ollama auto-setup + +When `RA_LLM__PROVIDER=ollama`, `ensure_setup()` runs health checks and may invoke `setups.manager.run_setup()` before the first query. Cloud providers skip this. + +## Progress output + +Live progress streams to **stderr** (stage bar, LLM previews). Disabled by `--no-progress`, `RA_PIPELINE__STREAM_PROGRESS=false`, or non-TTY stderr. + +Report content goes to **stdout** (or `--output` file). Redirect stderr to suppress progress in scripts: + +```bash +pipenv run python -m src --no-progress "query" 2>/dev/null +``` + +See [Progress streaming](../operations/progress-streaming.md). + +## Exit behavior + +- Setup failure → exit code `1` +- Retrieval exception in helper → error message printed, `None` returned +- Empty results (non-partial) → "no results" message + +## Programmatic alternative + +For full provider control from Python: + +```python +import asyncio +from src.config.settings import AppSettings +from src.retrieval.orchestrator import run_research + +async def main(): + settings = AppSettings( + retrieval={"providers": {"arxiv": {"enabled": True}}} + ) + report = await run_research("your query", settings=settings) + print(report.query) + +asyncio.run(main()) +``` + +See also: [CLI vs API](cli-vs-api.md), [Output formats](output-formats.md), [Configuration cookbook](configuration-cookbook.md), [Quick start](../getting-started/quick-start.md). diff --git a/docs/user-guide/configuration-cookbook.md b/docs/user-guide/configuration-cookbook.md new file mode 100644 index 0000000..d62dc31 --- /dev/null +++ b/docs/user-guide/configuration-cookbook.md @@ -0,0 +1,204 @@ +# Configuration Cookbook + +Copy-paste recipes for common setups. Each recipe shows **environment variables** and equivalent **YAML** where applicable. + +Precedence reminder: env > `.env` > YAML > defaults. See [Configuration precedence](../configuration/precedence.md). + +!!! tip "Pick your entry point first" + Batch CLI (`python -m src "query"`) only uses OpenAlex + Semantic Scholar regardless of provider recipes below. For recipes that enable arXiv/CrossRef or extra providers, use **interactive mode**, the **API**, or programmatic `run_research()`. See [CLI vs API](cli-vs-api.md). + +--- + +## 1. Fast local (default heuristic) + +Minimal config — Ollama auto model, heuristic synthesis, OpenAlex + S2 only on CLI batch path. + +**.env:** + +```bash +RA_LLM__PROVIDER=ollama +RA_LLM__MODEL=auto +RA_SYNTHESIS__LLM_ENABLED=false +RA_QUERY_EXPANSION__LLM_ENABLED=false +RA_PIPELINE__STREAM_PROGRESS=true +``` + +**Run:** [README Usage](https://github.com/Ndevu12/Research_Assistant_Model#usage) or [CLI reference](cli.md). + +**Best for:** Quick scans, low RAM (`llama3.2:3b`), offline-after-setup use. + +--- + +## 2. High-quality local (8B + LLM synthesis) + +Enable LLM stages when `llama3.1:8b` is selected (catalog sets `synthesis.llm_enabled: true` for 8B). + +**.env:** + +```bash +RA_LLM__PROVIDER=ollama +RA_LLM__MODEL=llama3.1:8b +RA_SYNTHESIS__LLM_ENABLED=true +RA_QUERY_EXPANSION__LLM_ENABLED=true +RA_SYNTHESIS__MAX_LLM_PAPERS=5 +``` + +**Verify model:** [Setup system](../setup-system/index.md#quick-start) (`health_check --model llama3.1:8b`). + +**Best for:** Richer synthesis on capable hardware (8–10 GB RAM). + +--- + +## 3. Cloud OpenAI quality + +Skip Ollama; use GPT for expansion and synthesis. + +**.env:** + +```bash +RA_LLM__PROVIDER=openai +RA_LLM__MODEL=gpt-4o-mini +OPENAI_API_KEY=sk-... +RA_SYNTHESIS__LLM_ENABLED=true +RA_QUERY_EXPANSION__LLM_ENABLED=true +RA_RANKING__TOP_K=30 +``` + +**Run:** [CLI reference](cli.md) (same query invocations as README Usage). + +**Best for:** Maximum quality without local GPU/RAM. API costs apply. + +--- + +## 4. Enable arXiv + CrossRef (full pipeline) + +!!! info "Not available on CLI batch shortcut" + Use **interactive mode**, **API**, or programmatic `run_research()` — not `python -m src "query"` alone. + +**YAML** (`config/providers.yaml`): + +```yaml +concurrency_limit: 4 +per_provider_limit: 10 +providers: + openalex: + enabled: true + semantic_scholar: + enabled: true + arxiv: + enabled: true + crossref: + enabled: true +``` + +**Equivalent .env:** + +```bash +RA_RETRIEVAL__PROVIDERS__ARXIV__ENABLED=true +RA_RETRIEVAL__PROVIDERS__CROSSREF__ENABLED=true +RA_RETRIEVAL__PER_PROVIDER_LIMIT=10 +RA_CROSSREF_MAILTO=you@example.com +S2_API_KEY=optional_for_rate_limits +``` + +**Run (interactive — full settings):** [Interactive sessions](interactive-sessions.md) (`pipenv run python -m src` with no query arg). + +Or use the [API](../api/index.md) / programmatic path. See [CLI vs API](cli-vs-api.md). + +--- + +## 5. Debug mode (inspect pipeline artifacts) + +**.env:** + +```bash +RA_PIPELINE__DEBUG=true +RA_PIPELINE__STREAM_PROGRESS=true +# Avoid duplicate: do not also set RA_DEBUG=1 unless intentional +``` + +**Run and inspect:** [Logging and debug walkthrough](../operations/logging-and-debug.md#debug-walkthrough), then: + +```bash +ls -lt logs/debug/ +jq '.stage_results | keys' logs/debug/pipeline_*.json +``` + +--- + +## 6. Retrieval-only smoke test + +Disable analysis/report stages to validate API connectivity quickly. + +**YAML snippet** (merge into `default.yaml` or use env): + +```yaml +pipeline: + enabled_stages: + query_understanding: true + query_expansion: true + retrieval: true + deduplication: true + ranking: true + relevance_scoring: false + clustering: false + synthesis: false + gap_analysis: false + citation_export: false + report_generation: false +``` + +**Env alternative (disable later stages):** + +```bash +RA_PIPELINE__ENABLED_STAGES__CLUSTERING=false +RA_PIPELINE__ENABLED_STAGES__SYNTHESIS=false +RA_PIPELINE__ENABLED_STAGES__GAP_ANALYSIS=false +RA_PIPELINE__ENABLED_STAGES__CITATION_EXPORT=false +RA_PIPELINE__ENABLED_STAGES__REPORT_GENERATION=false +``` + +Use debug dumps to inspect `artifacts.retrieved_papers`. + +--- + +## 7. Session memory + retrieval cache + +**.env:** + +```bash +RA_MEMORY__CACHE_ENABLED=true +RA_MEMORY__DB_PATH=data/research.db +``` + +**Interactive:** [Interactive sessions](interactive-sessions.md). + +Repeat the same query in a new session run may skip network retrieval when config hash matches. + +--- + +## 8. Quiet CI / scripting + +Disable progress via env and CLI flag — see [Progress streaming](../operations/progress-streaming.md#enabling-and-disabling) and [CLI reference](cli.md#progress-output). Example redirect: + +```bash +RA_PIPELINE__STREAM_PROGRESS=false +RA_PIPELINE__DEBUG=false +pipenv run python -m src --no-progress --format json "query" > report.json +``` + +--- + +## Recipe picker + +| Goal | Recipe | +|------|--------| +| Fastest offline run | #1 Fast local | +| Best local quality | #2 High-quality local | +| Best overall quality | #3 Cloud OpenAI | +| More paper sources | #4 arXiv + CrossRef | +| Diagnose failures | #5 Debug mode | +| Test APIs only | #6 Retrieval smoke test | +| Repeat queries faster | #7 Session cache | + +See also: [CLI vs API](cli-vs-api.md), [Environment variables](../configuration/environment-variables.md), [Heuristic vs LLM](../llm/heuristic-vs-llm.md), [Provider matrix](../retrieval/provider-matrix.md). diff --git a/docs/user-guide/interactive-sessions.md b/docs/user-guide/interactive-sessions.md new file mode 100644 index 0000000..22f0c2c --- /dev/null +++ b/docs/user-guide/interactive-sessions.md @@ -0,0 +1,162 @@ +# Interactive Sessions + +## Commands + +Start interactive mode: [README Usage — interactive](https://github.com/Ndevu12/Research_Assistant_Model#usage) or [CLI reference](cli.md). + +Session flags (`--format`, `--export`, `--no-progress`): [CLI reference — Arguments and flags](cli.md#arguments-and-flags). + +!!! info "Batch `--session` flag" + `python -m src --session "query"` is documented in the README but **not wired** in `__main__.py` as of current code. For SQLite memory and follow-ups, use interactive mode (no positional query). See [CLI vs API](cli-vs-api.md). + +## Architecture + +Implementation: `InteractiveResearchSession` in `src/memory/session.py`. + +```mermaid +flowchart TD + Start[Interactive loop __main__.py] --> Init[session.initialize] + Init --> Load[Load or create SQLite session] + Load --> Input[User input via get_user_query] + Input --> Parse[parse_follow_up filters.py] + Parse -->|NEW_QUERY| Full[run_full_query] + Parse -->|FILTER/FOCUS/COMPARE/EXPORT/HELP| FU[Session-side transform] + Full --> Pipe[run_research_with_result] + Pipe --> Store[MemoryStore persist] + FU --> Render[render_report_output] + Full --> Render + Render --> Print[stdout] + Print --> Input +``` + +### Startup sequence + +1. `run_interactive_mode()` prints welcome + `follow_up_help_text()` +2. Creates `InteractiveResearchSession()` with default `AppSettings()` +3. `initialize()` opens SQLite, loads existing session by ID or creates new +4. Restores `last_report` and `last_ranked_papers` from `session.context_json` if resuming + +### Full pipeline runs + +New research queries call `run_research_with_result()` with the **full** `AppSettings` merge — unlike batch CLI shortcut, **all enabled providers** in config are used. + +```python +# src/memory/session.py +report, pipeline_result = await run_research_with_result( + query, + settings=self.settings, # AppSettings() — not run_research_helper override + session=self.session, + store=self.store, +) +``` + +State persisted after each full run: + +- `last_report` — `EnhancedResearchReport` +- `last_ranked_papers` — extracted from `pipeline_result.artifacts["ranked_papers"]` +- Session metadata in SQLite (`data/research.db` by default) + +### Follow-up commands (no re-retrieval) + +After the first query, `parse_follow_up()` in `src/memory/filters.py` classifies input: + +| Intent | Trigger patterns | Handler | Network? | +|--------|------------------|---------|----------| +| `FILTER` | `papers after 2023`, `before 2020`, `from 2022`, `since 2019`, `filter: …` | `filter_ranked_papers()` + `filter_report_analyses()` | No | +| `FOCUS` | `focus on …`, `emphasize …` | `boost_ranked_papers()` (+0.15 per keyword hit) | No | +| `COMPARE` | `compare methodologies/datasets/benchmarks/approaches` | Markdown methodology list or full re-render | No | +| `EXPORT` | `export bibtex,apa,…` | `generate_citation_exports()` from cached papers | No | +| `HELP` | `help`, `?` | Returns `follow_up_help_text()` | No | +| `NEW_QUERY` | Anything else | Returns `None` → full pipeline re-run | Yes | + +Regex sources: `_AFTER_YEAR`, `_BEFORE_YEAR`, `_FROM_YEAR`, `_SINCE_YEAR`, `_FOCUS`, `_COMPARE`, `_EXPORT`, `_HELP` in `filters.py`. + +Help text (shown at session start): + +``` +Follow-up commands (no full re-retrieval): + • papers after 2023 / papers before 2020 / papers from 2022 + • focus on transformer approaches + • compare methodologies + • export bibtex,apa,mla,chicago +Enter a new research query for a full pipeline run. +``` + +### Follow-up rendering + +Follow-ups reuse the original pipeline's `partial` and `warnings` flags from `last_pipeline_result`. Filter/focus operations: + +1. Transform ranked papers locally +2. Build a filtered copy of `last_report` via `filter_report_analyses()` +3. Re-render with `render_report_output()` — same `--format` as session startup + +Focus updates prepend a note to `executive_summary`: *"Re-ranked by focus keywords (…)"*. + +## SQLite memory + +| Setting | Default | Env | +|---------|---------|-----| +| Database path | `data/research.db` | `RA_MEMORY__DB_PATH` | +| Retrieval cache | `false` | `RA_MEMORY__CACHE_ENABLED` | + +`MemoryStore` (`src/memory/store.py`) persists: + +- Session records (ID, context JSON, last query) +- Search history with cache keys +- Retrieved papers per search +- Rendered reports per format + +`context_json` stores serialized `last_report` and `last_ranked_papers` for follow-up commands across prompts within the same process. + +Enable retrieval caching to skip network when the same query + provider set + config hash was seen before: + +```bash +RA_MEMORY__CACHE_ENABLED=true +``` + +Cache lookup happens inside `run_research_with_result()` before pipeline execution. + +## Interactive vs batch CLI + +| Mode | Pipeline | Follow-ups | Memory | Providers | +|------|----------|------------|--------|-----------| +| Interactive (no query arg) | Full | Yes | Yes | Config-enabled | +| Batch `python -m src "q"` | Shortcut helper | No | No | OA + S2 only | + +Interactive mode is the primary way to use follow-up filters and full provider config from the CLI. + +## Output in sessions + +`--format` and `--export` passed at startup apply to full queries. Follow-up re-renders use the startup `--format`; export follow-ups use formats parsed from the `export …` command. + +JSON/HTML/PDF follow the same rules as batch mode. Interactive mode prints to stdout only — no `--output` file path in the session loop. + +See [Output formats](output-formats.md). + +## Tips + +- Run follow-ups only **after** a successful first query; otherwise: *"No previous research results in this session."* +- Focus and filter commands re-rank/filter locally — they cannot fetch new papers +- For provider changes (e.g. enable arXiv), set env/YAML **before** starting the session, then run a new query +- Use `Ctrl+C` or empty input (EOF) to exit — prints farewell message + +## Example session + +``` +$ pipenv run python -m src + +> transformer attention mechanisms +[progress on stderr…] +[markdown report on stdout] + +> papers after 2023 +[filtered report — no retrieval] + +> focus on sparse attention +[re-ranked report — no retrieval] + +> graph neural networks for molecules +[full pipeline re-run — new query] +``` + +See also: [CLI vs API](cli-vs-api.md), [CLI reference](cli.md), [Configuration — memory](../configuration/environment-variables.md), [Architecture — data model](../architecture/data-model.md). diff --git a/docs/user-guide/output-formats.md b/docs/user-guide/output-formats.md new file mode 100644 index 0000000..df047be --- /dev/null +++ b/docs/user-guide/output-formats.md @@ -0,0 +1,166 @@ +# Output Formats + + + +## Commands + +This page is the **canonical reference** for `--format`, `--export`, and the rendering pipeline. + +Copy-paste format commands: [README Output Format](https://github.com/Ndevu12/Research_Assistant_Model#output-format). CLI flags: [CLI reference](cli.md). + +## Rendering pipeline + +All CLI, interactive, and API paths converge on a single dispatcher: + +```mermaid +flowchart LR + R[EnhancedResearchReport] --> O[render_report_output] + O --> MD[render_enhanced_markdown] + O --> JS[render_json_report] + O --> HT[render_html_report] + HT --> PDF[pdf_ready CSS branch] + O --> EXP[generate_citation_exports] +``` + +Source: `src/reporting/output.py` + +```python +def render_report_output(report, *, output_format="markdown", partial=False, warnings=None, export_formats=None): + fmt = output_format.lower() + if fmt == "json": + return render_json_report(report, partial=partial, warnings=warnings or []) + if fmt in {"html", "pdf"}: + return render_html_report(..., pdf_ready=(fmt == "pdf")) + return render_enhanced_markdown(report, partial=partial, warnings=warnings, ...) +``` + +`export_formats` is accepted by the dispatcher signature but **citation exports are handled separately** in `run_research_helper()` and `InteractiveResearchSession._render_and_persist()` — they append export blocks after the main report rather than embedding them in markdown/HTML. + +## Format reference + +| `--format` | Renderer module | Output | Notes | +|------------|-----------------|--------|-------| +| `markdown` | `reporting/markdown.py` | stdout (default) | Thematic sections, clusters, synthesis, gaps | +| `json` | `reporting/json_report.py` | stdout or `--output` | Pydantic dump + `meta` block | +| `html` | `reporting/html.py` | file recommended | Standalone HTML with embedded CSS | +| `pdf` | `reporting/html.py` (`pdf_ready=True`) | file recommended | Print-ready HTML — **not** binary PDF | + +!!! info "PDF format" + `--format pdf` saves HTML optimized for printing. Open in a browser → **Print → Save as PDF**. The CLI prints a reminder when writing pdf output. + +### Markdown structure + +`render_enhanced_markdown()` builds these sections in order: + +1. **Pipeline notice** — when `partial=True` or `warnings` non-empty +2. **Query** and **Executive summary** +3. **Thematic findings** — one `## Theme N:` per cluster with linked papers +4. **Paper analyses** — per-paper key points, methodology, limitations +5. **Cross-paper synthesis** — agreements, disagreements, methodologies (when synthesis ran) +6. **Research gaps** — from `gap_analysis` +7. **Bibliography** — citation keys from `citation_index` + +Partial runs include warning banners at the top. Heuristic synthesis may show placeholder text when LLM stages were off. + +### JSON structure + +`report_to_dict()` wraps the pydantic model: + +```json +{ + "query": "...", + "executive_summary": "...", + "papers": [...], + "clusters": [...], + "synthesis": {...}, + "gap_analysis": {...}, + "citation_index": {...}, + "meta": { + "partial": false, + "warnings": [], + "paper_count": 12, + "cluster_count": 3 + } +} +``` + +Use JSON for downstream tooling, dashboards, or API integration. The API returns this structure in the `report` field plus a stringified copy in `rendered` when `format=json`. + +### HTML / PDF-ready + +`render_html_report()` produces a self-contained HTML document: + +- Semantic sections mirror markdown content +- Embedded CSS for screen reading +- When `pdf_ready=True`, adds `@media print` rules for page breaks and typography + +Always use `--output` for HTML/PDF — printing large HTML to the terminal triggers a tip message from `run_research_helper()`. + +## Citation exports (`--export`) + +Independent of `--format`. Supported values: + +| Format | Module | Description | +|--------|--------|-------------| +| `bibtex` | `src/export/` | BibTeX entries | +| `apa` | `src/export/` | APA 7th-style references | +| `mla` | `src/export/` | MLA-style references | +| `chicago` | `src/export/` | Chicago-style references | + +See [CLI reference — Examples](cli.md#examples) for `--export` usage. Supported values: `bibtex`, `apa`, `mla`, `chicago`. + +**Batch CLI flow** (`run_research_helper`): + +1. Render main report via `render_report_output()` +2. Write to `--output` if set, else print +3. Extract ranked papers from pipeline artifacts +4. Call `generate_citation_exports()` and print each format block after the report + +The pipeline also runs `CitationExportStage` internally for bibliography integration in the report body. + +**Interactive follow-up:** `export bibtex,apa` re-exports from cached ranked papers without re-retrieval. + +## Partial and warning metadata + +All renderers accept `partial` and `warnings` from `ResearchPipelineResult`: + +| Renderer | Behavior | +|----------|----------| +| Markdown | Notice section at top with warning list | +| HTML | `
` block | +| JSON | `meta.partial` and `meta.warnings` | + +Check stderr progress for ⚠ stage icons when output looks incomplete. + +## File output + +Use `--output` / `-o` with `--format html` or `--format pdf`. See [CLI reference — Examples](cli.md#examples) and [README Output Format](https://github.com/Ndevu12/Research_Assistant_Model#output-format). + +`run_research_helper()` creates parent directories automatically (`path.parent.mkdir(parents=True)`). + +Interactive mode always prints to stdout — there is no `--output` in the session loop. + +## API rendering + +FastAPI `POST /research` calls the same `render_report_output()`: + +| Request `format` | `rendered` field type | +|------------------|----------------------| +| `markdown` | string | +| `html` | string | +| `json` | string (JSON text) | + +The API does not support `pdf` format. The `export` request field is passed to `render_report_output()` but exports are primarily handled via the separate citation pipeline stage — see [API endpoints](../api/endpoints.md). + +## Entry point differences + +| Path | Who calls `render_report_output` | Export handling | +|------|----------------------------------|-----------------| +| CLI batch | `run_research_helper()` | Separate print after report | +| Interactive full query | `InteractiveResearchSession._render_and_persist()` | Same as batch | +| Interactive follow-up | `_render_local_report()` | `export …` command only | +| API | `app.py` route handler | Request `export` field | + +See [CLI vs API](cli-vs-api.md) for when each path runs. + +See also: [CLI reference](cli.md), [Report generation stage](../architecture/stages/report-generation.md), [Citation export stage](../architecture/stages/citation-export.md). diff --git a/mkdocs.yml b/mkdocs.yml new file mode 100644 index 0000000..db36ab0 --- /dev/null +++ b/mkdocs.yml @@ -0,0 +1,121 @@ +site_name: AI Research Assistant +site_url: https://ndevu12.github.io/Research_Assistant_Model/ +repo_url: https://github.com/Ndevu12/Research_Assistant_Model +edit_uri: edit/main/docs/ + +use_directory_urls: true + +exclude_docs: | + research-quality-known-issues.md + _analysis/* + +theme: + name: material + features: + - navigation.tabs + - navigation.sections + - navigation.expand + - search.suggest + - search.highlight + - content.code.copy + palette: + - media: "(prefers-color-scheme: light)" + scheme: default + primary: indigo + accent: indigo + toggle: + icon: material/brightness-7 + name: Switch to dark mode + - media: "(prefers-color-scheme: dark)" + scheme: slate + primary: indigo + accent: indigo + toggle: + icon: material/brightness-4 + name: Switch to light mode + +plugins: + - search + - mermaid2 + +markdown_extensions: + - admonition + - pymdownx.details + - pymdownx.superfences: + custom_fences: + - name: mermaid + class: mermaid + format: !!python/name:pymdownx.superfences.fence_code_format + - pymdownx.tabbed: + alternate_style: true + - tables + - toc: + permalink: true + +nav: + - Home: index.md + - Getting Started: + - getting-started/installation.md + - getting-started/quick-start.md + - getting-started/health-check.md + - User Guide: + - user-guide/cli.md + - user-guide/cli-vs-api.md + - user-guide/interactive-sessions.md + - user-guide/output-formats.md + - user-guide/configuration-cookbook.md + - Architecture: + - architecture/overview.md + - architecture/data-model.md + - architecture/pipeline-stages.md + - architecture/artifacts.md + - architecture/llm-layer.md + - Stages: + - architecture/stages/query-understanding.md + - architecture/stages/query-expansion.md + - architecture/stages/retrieval.md + - architecture/stages/deduplication.md + - architecture/stages/ranking.md + - architecture/stages/relevance-scoring.md + - architecture/stages/clustering.md + - architecture/stages/synthesis.md + - architecture/stages/gap-analysis.md + - architecture/stages/citation-export.md + - architecture/stages/report-generation.md + - Configuration: + - configuration/precedence.md + - configuration/environment-variables.md + - configuration/yaml-reference.md + - configuration/stage-toggles.md + - Retrieval: + - retrieval/overview.md + - retrieval/provider-matrix.md + - Providers: + - retrieval/per-provider/openalex.md + - retrieval/per-provider/semantic-scholar.md + - retrieval/per-provider/arxiv.md + - retrieval/per-provider/crossref.md + - LLM: + - llm/ollama.md + - llm/cloud-providers.md + - llm/heuristic-vs-llm.md + - API: + - api/index.md + - api/endpoints.md + - Operations: + - operations/logging-and-debug.md + - operations/progress-streaming.md + - operations/troubleshooting.md + - Development: + - development/local-setup.md + - development/testing.md + - development/import-conventions.md + - development/extensibility.md + - development/publishing.md + - contributing.md + - Reference: + - reference/canonical-sources.md + - Quality: + - quality/known-issues.md + - Setup System: + - setup-system/index.md diff --git a/scripts/check_docs_policy.py b/scripts/check_docs_policy.py new file mode 100755 index 0000000..9841bbf --- /dev/null +++ b/scripts/check_docs_policy.py @@ -0,0 +1,234 @@ +#!/usr/bin/env python3 +"""Docs policy checks for CI: scaffold placeholders, contributing pointer, deduplication.""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +DOCS = ROOT / "docs" +MKDOCS = ROOT / "mkdocs.yml" +CONTRIBUTING_POINTER = DOCS / "contributing.md" + +FENCED_BLOCK_RE = re.compile(r"^```[^\n]*\n(.*?)```", re.MULTILINE | re.DOTALL) +SRC_CMD_LINE_RE = re.compile(r"python -m src\b") + +CANONICAL_MARKERS: dict[str, str] = { + "setup-system/index.md": "", + "user-guide/cli.md": "", + "user-guide/output-formats.md": "", + "development/testing.md": "", + "operations/logging-and-debug.md": "", + "reference/canonical-sources.md": "", +} + +SETUP_COMMAND_ALLOWLIST = {"setup-system/index.md"} +MKDOCS_SERVE_ALLOWLIST = {"development/publishing.md"} +PYTEST_ALLOWLIST = { + "development/testing.md", + "quality/known-issues.md", + "development/extensibility.md", + "api/index.md", +} +SRC_CHEAT_SHEET_ALLOWLIST = { + "user-guide/cli.md", + "operations/logging-and-debug.md", + "operations/progress-streaming.md", + "user-guide/configuration-cookbook.md", + "quality/known-issues.md", + "user-guide/interactive-sessions.md", + "user-guide/cli-vs-api.md", + "api/index.md", +} +MIN_SRC_CMD_LINES_PER_BLOCK = 3 + +SCAFFOLD_MARKERS = ( + "", + "