diff --git a/.dockerignore b/.dockerignore index 68e5cab2da..9ebe303ba8 100644 --- a/.dockerignore +++ b/.dockerignore @@ -29,8 +29,12 @@ frontend/node_modules/ frontend/.next/ frontend/out/ -# Docs / non-runtime -docs/ +# Non-runtime. docs/ is NOT excluded: its markdown ships to /app/API/Docs and is indexed at +# runtime by the SearchDocs MCP tool (Get-CippDocsIndex). Only the image tree is dropped. +# This file is what applies to Dockerfile.release, which has no .dockerignore of +# its own — so excluding docs/ here breaks the release image's `COPY docs`, while leaving the +# dev image (which reads build/Dockerfile.dockerignore) working. Keep the two in step. +docs/.gitbook/ .env build/.env.example diff --git a/backend/Config/DocsPublishedPages.txt b/backend/Config/DocsPublishedPages.txt new file mode 100644 index 0000000000..b9d3e4eecd --- /dev/null +++ b/backend/Config/DocsPublishedPages.txt @@ -0,0 +1,431 @@ +# Slugs published on docs.cipp.app, snapshotted from llms.txt. +# Generated by build/tools/Update-DocsPublishedPages.ps1 - do not hand-edit. +# Read by Get-CippDocsPublishedSet so the docs search index never emits a URL that 404s. +# 427 pages. +api-documentation/endpoints +api-documentation/setup-and-authentication +demos/showcases +demos/tutorials +dev-documentation/cipp-dev-guide +dev-documentation/cipp-dev-guide/frontend-testing +dev-documentation/cipp-dev-guide/project-structure +dev-documentation/cipp-dev-guide/setting-up-for-local-development +dev-documentation/contributing-to-the-code +dev-documentation/contributing-to-the-documentation +msp-adoption-toolkit/implementing-cipp +msp-adoption-toolkit/implementing-cipp/msp-adoption-toolkit-building-a-cipp-business-case +msp-adoption-toolkit/implementing-cipp/why-cipp-doesnt-do-demos +msp-adoption-toolkit/sales-enablement-materials +msp-adoption-toolkit/sales-enablement-materials/content-templates-precooked +msp-adoption-toolkit/sales-enablement-materials/m365-management-package +msp-adoption-toolkit/sales-enablement-materials/security-packages +readme +security/cipp-community-vulnerability-disclosure-policy +security/cipp-security-and-compliance +security/cipp-security-and-compliance/security-policy +security/cipp-security-and-compliance/security-reports +setup/implementation-guide +setup/implementation-guide/recommended-first-steps +setup/implementation-guide/standards-setup +setup/installation +setup/installation/conditionalaccess +setup/installation/creating-the-cipp-service-account-gdap-ready +setup/installation/executing-the-setup-wizard +setup/installation/gdap-invite-wizard +setup/installation/owntenant +setup/maintaining-cipp +setup/maintaining-cipp/migrating-to-hosted-cipp +setup/maintaining-cipp/migrating-to-the-latest-version-of-cipp +setup/maintaining-cipp/recommended-roles +setup/maintaining-cipp/updating +setup/resources +setup/resources/how-cipp-evaluates-roles +setup/resources/professional-onboarding-services +setup/resources/sponsor-quick-start +setup/setting-up-cipp +setup/setting-up-cipp/customdomain +setup/setting-up-cipp/index +setup/setting-up-cipp/install +setup/setting-up-cipp/roles +sip-and-cipp/autopilot-and-intune +sip-and-cipp/conditional-access +sip-and-cipp/from-fork-to-feature +sip-and-cipp/rise-of-the-cipps +troubleshooting/frequently-asked-questions +troubleshooting/frequently-asked-questions/how-do-i-migrate-my-csp-to-a-new-tenant-in-cipp +troubleshooting/frequently-asked-questions/i-got-a-potential-phishing-page-detected-alert.-what-do-i-do-with-that +troubleshooting/frequently-asked-questions/standards-v-drift +troubleshooting/troubleshooting +troubleshooting/troubleshooting-instructions +troubleshooting/troubleshooting-instructions/refreshing-a-specific-tenants-permissions-via-cpv-api +troubleshooting/troubleshooting-instructions/repairing-missing-function-app-settings +user-documentation/cipp +user-documentation/cipp/advanced +user-documentation/cipp/advanced/authentication +user-documentation/cipp/advanced/authentication/cipp-roles +user-documentation/cipp/advanced/authentication/cipp-roles/add +user-documentation/cipp/advanced/authentication/cipp-users +user-documentation/cipp/advanced/authentication/sam-app-permissions +user-documentation/cipp/advanced/authentication/sam-app-roles +user-documentation/cipp/advanced/authentication/sso +user-documentation/cipp/advanced/container-management +user-documentation/cipp/advanced/container-management/custom-domains +user-documentation/cipp/advanced/container-management/logs +user-documentation/cipp/advanced/container-management/status +user-documentation/cipp/advanced/container-management/worker-health +user-documentation/cipp/advanced/diagnostics +user-documentation/cipp/advanced/exchange-cmdlets +user-documentation/cipp/advanced/super-admin +user-documentation/cipp/advanced/super-admin/function-offloading +user-documentation/cipp/advanced/super-admin/tenant-mode +user-documentation/cipp/advanced/super-admin/time-settings +user-documentation/cipp/advanced/table-maintenance +user-documentation/cipp/advanced/timers +user-documentation/cipp/custom-data +user-documentation/cipp/custom-data/directory-extensions +user-documentation/cipp/custom-data/directory-extensions/add +user-documentation/cipp/custom-data/mappings +user-documentation/cipp/custom-data/mappings/add +user-documentation/cipp/custom-data/mappings/edit +user-documentation/cipp/custom-data/schema-extensions +user-documentation/cipp/custom-data/schema-extensions/add +user-documentation/cipp/integrations +user-documentation/cipp/integrations/cipp-api +user-documentation/cipp/integrations/cloudflare +user-documentation/cipp/integrations/github +user-documentation/cipp/integrations/gradient +user-documentation/cipp/integrations/halopsa +user-documentation/cipp/integrations/have-i-been-pwned +user-documentation/cipp/integrations/hudu +user-documentation/cipp/integrations/integration-sync +user-documentation/cipp/integrations/ninjaone +user-documentation/cipp/integrations/passwordpusher +user-documentation/cipp/integrations/sherweb +user-documentation/cipp/logs +user-documentation/cipp/logs/logentry +user-documentation/cipp/sam-setup-wizard +user-documentation/cipp/settings +user-documentation/cipp/settings/backend +user-documentation/cipp/settings/backup +user-documentation/cipp/settings/branding +user-documentation/cipp/settings/features +user-documentation/cipp/settings/licenses +user-documentation/cipp/settings/notifications +user-documentation/cipp/settings/partner-webhooks +user-documentation/cipp/settings/password-config +user-documentation/cipp/settings/permissions +user-documentation/cipp/settings/siem +user-documentation/cipp/settings/tenants +user-documentation/copilot +user-documentation/copilot/agent365 +user-documentation/copilot/agent365/packages +user-documentation/copilot/reports +user-documentation/copilot/reports/copilot-adoption +user-documentation/copilot/reports/copilot-trend +user-documentation/copilot/reports/copilot-usage +user-documentation/copilot/settings +user-documentation/copilot/shadow-ai +user-documentation/dashboard +user-documentation/dashboard/custom +user-documentation/dashboard/dashboard +user-documentation/dashboard/devices +user-documentation/dashboard/identity +user-documentation/email +user-documentation/email/administration +user-documentation/email/administration/contacts +user-documentation/email/administration/contacts-template +user-documentation/email/administration/contacts-template/add +user-documentation/email/administration/contacts-template/edit +user-documentation/email/administration/contacts/edit +user-documentation/email/administration/deleted-mailboxes +user-documentation/email/administration/exchange-retention +user-documentation/email/administration/exchange-retention/policies +user-documentation/email/administration/exchange-retention/policies/policy +user-documentation/email/administration/exchange-retention/tags +user-documentation/email/administration/exchange-retention/tags/tag +user-documentation/email/administration/hve-accounts +user-documentation/email/administration/mailbox-rules +user-documentation/email/administration/mailboxes +user-documentation/email/administration/quarantine +user-documentation/email/administration/restricted-users +user-documentation/email/administration/tenant-allow-block-list-templates +user-documentation/email/administration/tenant-allow-block-lists +user-documentation/email/management +user-documentation/email/management/equipment +user-documentation/email/management/equipment/edit +user-documentation/email/management/list-rooms +user-documentation/email/management/list-rooms/edit +user-documentation/email/management/room-lists +user-documentation/email/management/room-lists/edit +user-documentation/email/reports +user-documentation/email/reports/activesync-devices +user-documentation/email/reports/antiphishing-filters +user-documentation/email/reports/calendar-permissions +user-documentation/email/reports/global-address-list +user-documentation/email/reports/mailbox-activity +user-documentation/email/reports/mailbox-cas-settings +user-documentation/email/reports/mailbox-forwarding +user-documentation/email/reports/mailbox-permissions +user-documentation/email/reports/mailbox-statistics +user-documentation/email/reports/malware-filters +user-documentation/email/reports/safeattachments-filters +user-documentation/email/reports/sharedmailboxenabledaccount +user-documentation/email/spamfilter +user-documentation/email/spamfilter/list-connectionfilter +user-documentation/email/spamfilter/list-connectionfilter-templates +user-documentation/email/spamfilter/list-connectionfilter/add +user-documentation/email/spamfilter/list-quarantine-policies +user-documentation/email/spamfilter/list-quarantine-policies/add +user-documentation/email/spamfilter/list-spamfilter +user-documentation/email/spamfilter/list-spamfilter/add +user-documentation/email/spamfilter/list-templates +user-documentation/email/transport +user-documentation/email/transport/list-connector-templates +user-documentation/email/transport/list-connectors +user-documentation/email/transport/list-rules +user-documentation/email/transport/list-templates +user-documentation/endpoint +user-documentation/endpoint/applications +user-documentation/endpoint/applications/application-templates +user-documentation/endpoint/applications/list +user-documentation/endpoint/applications/queue +user-documentation/endpoint/autopilot +user-documentation/endpoint/autopilot/add-device +user-documentation/endpoint/autopilot/enrollment-profiles +user-documentation/endpoint/autopilot/enrollment-profiles/android-enterprise +user-documentation/endpoint/autopilot/enrollment-profiles/apple-ade +user-documentation/endpoint/autopilot/list-devices +user-documentation/endpoint/autopilot/list-status-pages +user-documentation/endpoint/mem +user-documentation/endpoint/mem/approval-requests +user-documentation/endpoint/mem/assignment-filter-templates +user-documentation/endpoint/mem/assignment-filter-templates/add +user-documentation/endpoint/mem/assignment-filter-templates/deploy +user-documentation/endpoint/mem/assignment-filter-templates/edit-assignment-filter-template +user-documentation/endpoint/mem/assignment-filters +user-documentation/endpoint/mem/assignment-filters/add +user-documentation/endpoint/mem/assignment-filters/edit +user-documentation/endpoint/mem/bitlocker-search +user-documentation/endpoint/mem/devices +user-documentation/endpoint/mem/devices/device +user-documentation/endpoint/mem/list-appprotection-policies +user-documentation/endpoint/mem/list-compliance-policies +user-documentation/endpoint/mem/list-policies +user-documentation/endpoint/mem/list-scripts +user-documentation/endpoint/mem/list-templates +user-documentation/endpoint/mem/list-templates/edit +user-documentation/endpoint/mem/reusable-settings +user-documentation/endpoint/mem/reusable-settings-templates +user-documentation/endpoint/mem/reusable-settings-templates/add +user-documentation/endpoint/mem/reusable-settings-templates/edit-reusable-settings-template +user-documentation/endpoint/mem/reusable-settings/edit +user-documentation/endpoint/reports +user-documentation/endpoint/reports/analyticsdevicescore +user-documentation/endpoint/reports/autopilot-deployment +user-documentation/endpoint/reports/detected-apps +user-documentation/endpoint/reports/work-from-anywhere +user-documentation/identity +user-documentation/identity/administration +user-documentation/identity/administration/deleted-items +user-documentation/identity/administration/devices +user-documentation/identity/administration/group-templates +user-documentation/identity/administration/group-templates/add +user-documentation/identity/administration/group-templates/deploy +user-documentation/identity/administration/group-templates/edit +user-documentation/identity/administration/groups +user-documentation/identity/administration/groups/add +user-documentation/identity/administration/groups/edit +user-documentation/identity/administration/groups/group +user-documentation/identity/administration/jit-admin +user-documentation/identity/administration/jit-admin-templates +user-documentation/identity/administration/jit-admin-templates/add-jit-admin-template +user-documentation/identity/administration/jit-admin-templates/edit-jit-admin-template +user-documentation/identity/administration/jit-admin/add +user-documentation/identity/administration/offboarding-wizard +user-documentation/identity/administration/risky-users +user-documentation/identity/administration/roles +user-documentation/identity/administration/users +user-documentation/identity/administration/users/patch-wizard +user-documentation/identity/administration/users/user +user-documentation/identity/administration/users/user/bec +user-documentation/identity/administration/users/user/conditional-access +user-documentation/identity/administration/users/user/edit +user-documentation/identity/administration/users/user/exchange +user-documentation/identity/administration/vacation-mode +user-documentation/identity/administration/vacation-mode/add-vacation-schedule +user-documentation/identity/reports +user-documentation/identity/reports/azure-ad-connect-report +user-documentation/identity/reports/inactive-users-report +user-documentation/identity/reports/mfa-report +user-documentation/identity/reports/risk-detections +user-documentation/identity/reports/signin-report +user-documentation/security +user-documentation/security/compliance +user-documentation/security/compliance/dlp +user-documentation/security/compliance/dlp-templates +user-documentation/security/compliance/labels +user-documentation/security/compliance/labels-templates +user-documentation/security/compliance/retention +user-documentation/security/compliance/retention-templates +user-documentation/security/compliance/sit +user-documentation/security/compliance/sit-templates +user-documentation/security/defender +user-documentation/security/defender/defender-cve-exceptions +user-documentation/security/defender/deployment +user-documentation/security/defender/list-defender +user-documentation/security/defender/list-defender-tvm +user-documentation/security/incidents +user-documentation/security/incidents/list-alerts +user-documentation/security/incidents/list-check-alerts +user-documentation/security/incidents/list-incidents +user-documentation/security/incidents/list-mdo-alerts +user-documentation/security/reports +user-documentation/security/reports/cve-report +user-documentation/security/reports/list-device-compliance +user-documentation/security/reports/mde-onboarding +user-documentation/security/safelinks +user-documentation/security/safelinks/safelinks +user-documentation/security/safelinks/safelinks-template +user-documentation/security/safelinks/safelinks-template/add +user-documentation/security/safelinks/safelinks-template/create +user-documentation/security/safelinks/safelinks-template/edit +user-documentation/security/safelinks/safelinks/add +user-documentation/security/safelinks/safelinks/edit +user-documentation/shared-features +user-documentation/shared-features/breadcrumb-navigation +user-documentation/shared-features/get-help +user-documentation/shared-features/global-page-icon +user-documentation/shared-features/keyboard-shortcuts +user-documentation/shared-features/menu-bar +user-documentation/shared-features/menu-bar/bookmarks +user-documentation/shared-features/menu-bar/display-mode +user-documentation/shared-features/menu-bar/search +user-documentation/shared-features/menu-bar/tenant-select +user-documentation/shared-features/menu-bar/universal-search +user-documentation/shared-features/menu-bar/user-settings +user-documentation/shared-features/release-notes-notification +user-documentation/shared-features/speed-dial +user-documentation/shared-features/table-features +user-documentation/shared-features/variable-auto-complete +user-documentation/teams-share +user-documentation/teams-share/deleted-sites +user-documentation/teams-share/external-users +user-documentation/teams-share/onedrive +user-documentation/teams-share/permissions-report +user-documentation/teams-share/sharepoint +user-documentation/teams-share/sharepoint-templates +user-documentation/teams-share/sharepoint-templates/add +user-documentation/teams-share/sharepoint/add-site +user-documentation/teams-share/sharepoint/bulk-add-site +user-documentation/teams-share/sharing-report +user-documentation/teams-share/teams +user-documentation/teams-share/teams/business-voice +user-documentation/teams-share/teams/list-team +user-documentation/teams-share/teams/list-team/add +user-documentation/teams-share/teams/teams-activity +user-documentation/tenant +user-documentation/tenant/administration +user-documentation/tenant/administration/alert-configuration +user-documentation/tenant/administration/alert-configuration/alert +user-documentation/tenant/administration/alert-configuration/snoozed-alerts +user-documentation/tenant/administration/app-consent-requests +user-documentation/tenant/administration/applications +user-documentation/tenant/administration/applications/app-registrations +user-documentation/tenant/administration/applications/app-registrations/appid +user-documentation/tenant/administration/applications/enterprise-apps +user-documentation/tenant/administration/applications/enterprise-apps/spid +user-documentation/tenant/administration/applications/permission-sets +user-documentation/tenant/administration/applications/templates +user-documentation/tenant/administration/applications/templates/add +user-documentation/tenant/administration/applications/templates/edit +user-documentation/tenant/administration/audit-logs +user-documentation/tenant/administration/audit-logs/directory-audits +user-documentation/tenant/administration/audit-logs/log +user-documentation/tenant/administration/audit-logs/manual-searches +user-documentation/tenant/administration/audit-logs/search-results +user-documentation/tenant/administration/audit-logs/searches +user-documentation/tenant/administration/authentication-methods +user-documentation/tenant/administration/authentication-methods/registration-campaign +user-documentation/tenant/administration/domains +user-documentation/tenant/administration/partner-relationships +user-documentation/tenant/administration/securescore +user-documentation/tenant/administration/securescore/table +user-documentation/tenant/administration/tenants +user-documentation/tenant/administration/tenants/global-variables +user-documentation/tenant/administration/tenants/groups +user-documentation/tenant/administration/tenants/groups/edit +user-documentation/tenant/conditional +user-documentation/tenant/conditional/list-named-locations +user-documentation/tenant/conditional/list-named-locations/add +user-documentation/tenant/conditional/list-policies +user-documentation/tenant/conditional/list-policies/edit-ca-policy +user-documentation/tenant/conditional/list-template +user-documentation/tenant/conditional/list-template/create-ca-template +user-documentation/tenant/conditional/list-template/edit +user-documentation/tenant/gdap-management +user-documentation/tenant/gdap-management/invites +user-documentation/tenant/gdap-management/invites/add +user-documentation/tenant/gdap-management/offboarding +user-documentation/tenant/gdap-management/onboarding +user-documentation/tenant/gdap-management/onboarding/start +user-documentation/tenant/gdap-management/relationships +user-documentation/tenant/gdap-management/relationships/relationship +user-documentation/tenant/gdap-management/relationships/relationship/mappings +user-documentation/tenant/gdap-management/role-templates +user-documentation/tenant/gdap-management/role-templates/add +user-documentation/tenant/gdap-management/role-templates/edit +user-documentation/tenant/gdap-management/roles +user-documentation/tenant/gdap-management/roles/add +user-documentation/tenant/manage +user-documentation/tenant/manage/applied-standards +user-documentation/tenant/manage/backup +user-documentation/tenant/manage/drift +user-documentation/tenant/manage/edit +user-documentation/tenant/manage/history +user-documentation/tenant/manage/policies-deployed +user-documentation/tenant/manage/user-defaults +user-documentation/tenant/reports +user-documentation/tenant/reports/application-consent +user-documentation/tenant/reports/custom-test-report +user-documentation/tenant/reports/graph-office-reports +user-documentation/tenant/reports/list-csp-licenses +user-documentation/tenant/reports/list-csp-licenses/add-subscription +user-documentation/tenant/reports/list-licenses +user-documentation/tenant/standards +user-documentation/tenant/standards/alignment +user-documentation/tenant/standards/alignment/templates +user-documentation/tenant/standards/alignment/templates/available-standards +user-documentation/tenant/standards/bpa-report +user-documentation/tenant/standards/bpa-report/best-practice-templates +user-documentation/tenant/standards/bpa-report/builder +user-documentation/tenant/standards/domains-analyser +user-documentation/tenant/standards/domains-analyser/domain-analyser-updates-and-data-refreshing +user-documentation/tenant/standards/template +user-documentation/tools +user-documentation/tools/community-repos +user-documentation/tools/community-repos/browse-all-templates +user-documentation/tools/custom-tests +user-documentation/tools/custom-tests/add +user-documentation/tools/custom-tests/versions +user-documentation/tools/dark-web-tools +user-documentation/tools/dark-web-tools/breach-lookup +user-documentation/tools/dark-web-tools/tenant-breach-lookup +user-documentation/tools/email-tools +user-documentation/tools/email-tools/mailbox-restores +user-documentation/tools/email-tools/message-trace +user-documentation/tools/email-tools/message-viewer +user-documentation/tools/intune-tools +user-documentation/tools/intune-tools/compare-policies +user-documentation/tools/report-builder +user-documentation/tools/report-builder/builder +user-documentation/tools/report-builder/generated +user-documentation/tools/report-builder/templates +user-documentation/tools/scheduler +user-documentation/tools/scheduler/task +user-documentation/tools/templatelib +user-documentation/tools/tenant-tools +user-documentation/tools/tenant-tools/appapproval +user-documentation/tools/tenant-tools/geoiplookup +user-documentation/tools/tenant-tools/graph-explorer +user-documentation/tools/tenant-tools/individual-domains +user-documentation/tools/tenant-tools/tenantlookup diff --git a/backend/Config/DocsSynonyms.json b/backend/Config/DocsSynonyms.json new file mode 100644 index 0000000000..91affbe50f --- /dev/null +++ b/backend/Config/DocsSynonyms.json @@ -0,0 +1,75 @@ +{ + "_comment": "Query-expansion map for the SearchDocs MCP tool (Find-CippDoc). Keys are matched against the caller's query after tokenisation and stemming, so write them as ordinary words; the loader stems them. Values are expansion phrases scored at a damped weight, which is what lets a search for 'CA policy' reach pages that only ever say 'conditional access'. This is where most of the perceived semantic behaviour comes from without an embedding model, so it is worth extending whenever a real search misses.", + "expansions": { + "ca": ["conditional access policy"], + "mfa": ["multifactor authentication", "multi factor"], + "2fa": ["multifactor authentication"], + "sso": ["single sign on", "saml", "identity provider"], + "gdap": ["granular delegated admin privileges", "delegated access", "relationship"], + "dap": ["delegated admin privileges"], + "sam": ["secure application model", "service account", "application registration"], + "bec": ["business email compromise", "compromise remediation", "indicators of compromise"], + "bpa": ["best practice analyser", "report builder"], + "cis": ["compliance benchmark test"], + "spf": ["domain analyser", "email authentication", "dns record"], + "dkim": ["domain analyser", "email authentication", "dns record"], + "dmarc": ["domain analyser", "email authentication", "dns record"], + "dns": ["domain analyser", "domain health"], + "offboard": ["offboarding user removal"], + "onboard": ["onboarding tenant setup wizard"], + "standard": ["drift remediation baseline template"], + "drift": ["standards deviation baseline"], + "alert": ["alerting notification webhook"], + "tenant": ["customer client organisation"], + "intune": ["endpoint manager device management"], + "autopilot": ["device enrolment provisioning"], + "defender": ["security threat protection antivirus"], + "exchange": ["exchange online mailbox email"], + "exo": ["exchange online"], + "spam": ["spamfilter quarantine mail flow"], + "quarantine": ["spamfilter released message"], + "mailbox": ["exchange mailbox permissions shared"], + "license": ["licence sku subscription assignment"], + "licence": ["license sku subscription assignment"], + "sku": ["license subscription"], + "role": ["permission access rbac custom role"], + "permission": ["role access rbac consent"], + "rbac": ["role based access control permission"], + "log": ["audit log logbook activity history"], + "audit": ["log logbook activity history"], + "webhook": ["notification alert subscription"], + "psa": ["integration halo autotask connectwise"], + "rmm": ["integration ninja datto syncro"], + "backup": ["restore recovery export"], + "restore": ["backup recovery import"], + "template": ["policy blueprint preset"], + "policy": ["template configuration profile"], + "group": ["distribution list security group team"], + "user": ["account identity member"], + "password": ["credential reset passwordless authentication method"], + "device": ["endpoint computer workstation managed device"], + "app": ["application enterprise application service principal"], + "application": ["app enterprise application service principal"], + "sharepoint": ["onedrive site document library"], + "onedrive": ["sharepoint site storage"], + "teams": ["team channel meeting collaboration"], + "report": ["reporting export dashboard analytics"], + "dashboard": ["overview home report"], + "scheduler": ["scheduled task recurring job cron"], + "queue": ["scheduled task job processing"], + "error": ["troubleshooting failure issue problem"], + "fail": ["troubleshooting error issue problem"], + "troubleshoot": ["error failure diagnostic issue"], + "install": ["deployment setup provisioning"], + "deploy": ["installation setup provisioning"], + "upgrade": ["update version migration"], + "update": ["upgrade version release"], + "api": ["endpoint integration rest client"], + "mcp": ["model context protocol tool integration"], + "copilot": ["microsoft copilot ai"], + "hosted": ["cyberdrain hosted managed instance sponsor"], + "selfhost": ["self hosted azure deployment"], + "azure": ["subscription resource group function app"], + "graph": ["microsoft graph api request"] + } +} diff --git a/backend/Config/openapi.json b/backend/Config/openapi.json index 2f995bd299..454444db4e 100644 --- a/backend/Config/openapi.json +++ b/backend/Config/openapi.json @@ -4670,10 +4670,20 @@ } }, "TemplateGuid": { - "type": "string" + "type": "string", + "description": "The deploy drawer and wizard send the chosen row's GUID as TemplateList.value, not as TemplateID. Template display names are not unique - re-imports create same-named twins - so resolving by display name below can land on a different row than the one the user picked. The selected RowKey must win whenever the request carries one. String rather than Guid: built-in templates are stored with their filename as RowKey." }, "TemplateID": { - "type": "string" + "type": "string", + "description": "The deploy drawer and wizard send the chosen row's GUID as TemplateList.value, not as TemplateID. Template display names are not unique - re-imports create same-named twins - so resolving by display name below can land on a different row than the one the user picked. The selected RowKey must win whenever the request carries one. String rather than Guid: built-in templates are stored with their filename as RowKey." + }, + "TemplateList": { + "allOf": [ + { + "$ref": "#/components/schemas/LabelValue" + } + ], + "description": "The deploy drawer and wizard send the chosen row's GUID as TemplateList.value, not as TemplateID. Template display names are not unique - re-imports create same-named twins - so resolving by display name below can land on a different row than the one the user picked. The selected RowKey must win whenever the request carries one. String rather than Guid: built-in templates are stored with their filename as RowKey." }, "TemplateType": { "type": "string" @@ -38916,6 +38926,86 @@ "x-cipp-role": "CIPP.Core.Read" } }, + "/api/ListCippDocs": { + "get": { + "summary": "Search the CIPP documentation, or fetch one documentation page in full.", + "operationId": "ListCippDocs", + "tags": [ + "CIPP > Core" + ], + "description": "Searches the GitBook documentation shipped with this build and returns matching sections,\neach with an excerpt and links back to docs.cipp.app and to the file on GitHub. Pages under\nuser-documentation also report the CIPP route they document, so a screen can be traced to\nits docs and back.\n\nPass path on its own to list the pages under a documentation subtree or a CIPP route, or\nwith full=true to return one page's entire text. This backs the SearchDocs and GetDoc MCP\ntools and is available to the UI and API clients on the same terms.", + "parameters": [ + { + "name": "full", + "in": "query", + "description": "Return the whole page rather than matching sections. Requires path.", + "required": false, + "schema": { + "type": "string" + } + }, + { + "name": "limit", + "in": "query", + "description": "Maximum results to return (default 8, max 25).", + "required": false, + "schema": { + "type": "string" + } + }, + { + "name": "path", + "in": "query", + "description": "A documentation subtree ('user-documentation/identity') or a CIPP route ('/identity/administration/users').", + "required": true, + "schema": { + "type": "string" + } + }, + { + "name": "query", + "in": "query", + "description": "Keywords or a plain-language question, e.g. 'how do I set up GDAP'.", + "required": false, + "schema": { + "type": "string" + } + } + ], + "responses": { + "200": { + "description": "Success", + "content": { + "application/json": { + "schema": { + "type": "array", + "items": { + "type": "object", + "description": "Not described statically: this endpoint returns the upstream response as-is, so its fields are determined by the upstream API rather than by CIPP. Call the endpoint to see the actual shape, or add a response schema in backend/Config/openapi-overrides." + } + } + } + } + }, + "401": { + "description": "Unauthorized - invalid or missing bearer token" + }, + "403": { + "description": "Forbidden - caller lacks the required RBAC role" + }, + "500": { + "description": "Internal server error" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "x-cipp-role": "CIPP.Core.Read", + "x-cipp-any-tenant": true + } + }, "/api/ListCippQueue": { "get": { "summary": "ListCippQueue", @@ -46309,21 +46399,24 @@ "type": "object", "description": "Derived from the fields written into the storage table it reads, and the fields the endpoint selects onto each record, and the columns the CIPP UI renders. Fields taken from the storage writers may be omitted by this endpoint, and the response may carry computed fields not listed here.", "properties": { + "corrupt": { + "x-cipp-field-source": "backend" + }, "description": { - "x-cipp-field-source": "frontend" + "x-cipp-field-source": "backend,frontend" }, "displayName": { - "x-cipp-field-source": "frontend" + "x-cipp-field-source": "backend,frontend" }, "ETag": { "type": "string", "x-cipp-field-source": "storage" }, "guid": { - "x-cipp-field-source": "storage" + "x-cipp-field-source": "storage,backend" }, "isSynced": { - "x-cipp-field-source": "frontend" + "x-cipp-field-source": "backend,frontend" }, "JSON": { "x-cipp-field-source": "storage" @@ -46332,7 +46425,7 @@ "x-cipp-field-source": "backend" }, "package": { - "x-cipp-field-source": "storage,frontend" + "x-cipp-field-source": "storage,backend,frontend" }, "PartitionKey": { "x-cipp-field-source": "storage" @@ -46348,7 +46441,7 @@ "x-cipp-field-source": "storage" }, "source": { - "x-cipp-field-source": "storage" + "x-cipp-field-source": "storage,backend" }, "templateCount": { "x-cipp-field-source": "backend" @@ -49144,6 +49237,48 @@ "x-cipp-role": "Tenant.Relationship.Read" } }, + "/api/ListPartnerTenantInfo": { + "get": { + "summary": "ListPartnerTenantInfo", + "operationId": "ListPartnerTenantInfo", + "tags": [ + "CIPP > Core" + ], + "description": "Reports whether the CIPP host tenant is a Microsoft Partner tenant, so the frontend can\ndecide whether partner-only flows (GDAP onboarding, reseller invites, GDAP permission\nchecks) apply to this instance.\n\nMarked AnyTenant deliberately. This answers a question about the CIPP instance, not\nabout a tenant the caller wants to act on, and Get-CippPartnerTenantInfo pins the lookup\nto $env:TenantID. Without the flag, Test-CIPPAccess falls back to $env:TenantID as the\ntenant filter and denies any custom role that blocks the partner tenant, which silently\ngreys out partner-only UI for roles that are otherwise fully permitted.", + "responses": { + "200": { + "description": "Success", + "content": { + "application/json": { + "schema": { + "type": "array", + "items": { + "type": "object", + "description": "Not described statically: this endpoint returns the upstream response as-is, so its fields are determined by the upstream API rather than by CIPP. Call the endpoint to see the actual shape, or add a response schema in backend/Config/openapi-overrides." + } + } + } + } + }, + "401": { + "description": "Unauthorized - invalid or missing bearer token" + }, + "403": { + "description": "Forbidden - caller lacks the required RBAC role" + }, + "500": { + "description": "Internal server error" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "x-cipp-role": "CIPP.Core.Read", + "x-cipp-any-tenant": true + } + }, "/api/ListPendingWebhooks": { "get": { "summary": "ListPendingWebhooks", diff --git a/backend/Modules/CIPPCore/Public/MCP/ConvertFrom-CippDocMarkdown.ps1 b/backend/Modules/CIPPCore/Public/MCP/ConvertFrom-CippDocMarkdown.ps1 new file mode 100644 index 0000000000..8ca637c9c5 --- /dev/null +++ b/backend/Modules/CIPPCore/Public/MCP/ConvertFrom-CippDocMarkdown.ps1 @@ -0,0 +1,122 @@ +function ConvertFrom-CippDocMarkdown { + <# + .SYNOPSIS + Parses a GitBook markdown page into a title, description and heading-delimited chunks. + .DESCRIPTION + Strips the machinery GitBook layers on top of markdown so it does not end up in the search + index or in a returned snippet: YAML frontmatter (the description is kept), '{% ... %}' + block tags such as {% stepper %} / {% hint %}, raw
//
embeds, and the + markdown link/emphasis syntax. Link labels survive because they are real prose; the URLs + do not, because a query should not match a page on the strength of a href. + + Splits at '##' and '###' headings. Text before the first heading becomes the intro chunk, + which is what a query about the page as a whole should match. Fenced code blocks are kept + as text - PowerShell examples in the docs are frequently the thing being searched for - but + their fence markers and language hints are dropped. Headings inside a fence are ignored, so + a '# comment' in a shell example does not split the page. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param( + [Parameter(Mandatory)][AllowEmptyString()][string]$Markdown, + [Parameter(Mandatory)][string]$RelativePath + ) + + $Description = '' + $Body = $Markdown + + # YAML frontmatter, only when it opens the file. + if ($Body -match '(?s)^?---\r?\n(.*?)\r?\n---\r?\n?(.*)$') { + $FrontMatter = $Matches[1] + $Body = $Matches[2] + if ($FrontMatter -match '(?m)^description:\s*(.+?)\s*$') { + $Description = $Matches[1].Trim().Trim('"', "'") + } + } + + # GitBook block tags: {% stepper %}, {% hint style="info" %}, {% endstep %}, {% include ... %}. + $Body = $Body -replace '(?s)\{%.*?%\}', ' ' + # Raw HTML embeds - figures, images and the sponsor
grids - carry no searchable prose. + $Body = $Body -replace '(?s)
.*?
', ' ' + $Body = $Body -replace '(?s)<(script|style)\b.*?', ' ' + $Body = $Body -replace '<[^>]+>', ' ' + # Images before links, so an image's alt text does not survive as if it were a link label. + $Body = $Body -replace '!\[[^\]]*\]\([^)]*\)', ' ' + $Body = $Body -replace '\[([^\]]*)\]\([^)]*\)', '$1' + $Body = $Body -replace ' ', ' ' + + $Title = '' + $Chunks = [System.Collections.Generic.List[object]]::new() + + $CurrentHeading = '' + $CurrentText = [System.Text.StringBuilder]::new() + $InFence = $false + + $AddChunk = { + $Text = $CurrentText.ToString() + $Text = ($Text -replace '[ \t]+', ' ' -replace '(\r?\n\s*){2,}', "`n").Trim() + if ($Text -or $CurrentHeading) { + $Chunks.Add([pscustomobject]@{ Heading = $CurrentHeading; Text = $Text }) + } + } + + foreach ($Line in ($Body -split '\r?\n')) { + if ($Line -match '^\s*(```|~~~)') { + $InFence = -not $InFence + continue + } + + if (-not $InFence -and $Line -match '^(#{1,6})\s+(.*\S)\s*$') { + $Level = $Matches[1].Length + $Text = ($Matches[2] -replace '[`*_~]', '').Trim() + + if ($Level -eq 1 -and -not $Title) { + # The page's own H1 titles the page; it does not start a chunk. + $Title = $Text + continue + } + + if ($Level -le 3) { + & $AddChunk + $CurrentHeading = $Text + $CurrentText = [System.Text.StringBuilder]::new() + continue + } + # H4+ stays inside the current chunk as ordinary emphasised prose. + [void]$CurrentText.AppendLine($Text) + continue + } + + [void]$CurrentText.AppendLine($Line) + } + & $AddChunk + + if (-not $Title) { + # No H1: fall back to the file (or folder, for a README) name, title-cased. + $Leaf = [System.IO.Path]::GetFileNameWithoutExtension($RelativePath) + if ($Leaf -match '(?i)^README$') { + $Parent = ($RelativePath -replace '\\', '/') -replace '/[^/]+$', '' + $Leaf = ($Parent -split '/')[-1] + } + $Title = (($Leaf -replace '[-_]', ' ') -split ' ' | Where-Object { $_ } | ForEach-Object { + $_.Substring(0, 1).ToUpperInvariant() + $_.Substring(1) + }) -join ' ' + } + + # Breadcrumb from the folder chain, for display: 'User Documentation > Identity > Users'. + $Segments = @(($RelativePath -replace '\\', '/') -split '/') + $Folders = @($Segments | Select-Object -SkipLast 1) + $Breadcrumb = (@($Folders | ForEach-Object { + (($_ -replace '[-_]', ' ') -split ' ' | Where-Object { $_ } | ForEach-Object { + $_.Substring(0, 1).ToUpperInvariant() + $_.Substring(1) + }) -join ' ' + }) -join ' > ') + + return [pscustomobject]@{ + Title = $Title + Description = $Description + Breadcrumb = $Breadcrumb + Chunks = $Chunks + } +} diff --git a/backend/Modules/CIPPCore/Public/MCP/ConvertTo-CippDocToken.ps1 b/backend/Modules/CIPPCore/Public/MCP/ConvertTo-CippDocToken.ps1 new file mode 100644 index 0000000000..8738523156 --- /dev/null +++ b/backend/Modules/CIPPCore/Public/MCP/ConvertTo-CippDocToken.ps1 @@ -0,0 +1,20 @@ +function ConvertTo-CippDocToken { + <# + .SYNOPSIS + Tokenises text for the docs index: lowercase, split, de-stopped, lightly stemmed. + .DESCRIPTION + A thin wrapper over CIPPSharp's tokeniser, which is the single implementation. Indexing + and querying must tokenise identically - a term indexed as 'standard' is never found by a + query for 'Standards' otherwise - so there is deliberately no second copy of these rules + in PowerShell. The rules themselves are documented on CIPP.DocsIndex.Tokenize. + + This exists so callers and tests can tokenise without reaching for the type name, and so + the query path reads the same as the rest of the MCP code. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param([AllowEmptyString()][string]$Text) + + return @([CIPP.DocsIndex]::Tokenize($Text)) +} diff --git a/backend/Modules/CIPPCore/Public/MCP/Find-CippDoc.ps1 b/backend/Modules/CIPPCore/Public/MCP/Find-CippDoc.ps1 new file mode 100644 index 0000000000..9b431e70c3 --- /dev/null +++ b/backend/Modules/CIPPCore/Public/MCP/Find-CippDoc.ps1 @@ -0,0 +1,218 @@ +function Find-CippDoc { + <# + .SYNOPSIS + Searches the CIPP documentation and returns ranked, deep-linked sections. + .DESCRIPTION + Backs the SearchDocs core tool. Scoring is BM25 over CIPPSharp's inverted index, with + three expansions layered on top of the caller's literal terms. None of them is a vector + model - CIPP has no embedding provider, and requiring an API key to search the docs would + be a poor trade - but together they cover most of what a caller means rather than types: + + - Domain synonyms (Config/DocsSynonyms.json), damped to 0.55. This is the one that + matters: 'CA policy' reaches pages that only ever write 'conditional access'. + - Fuzzy vocabulary matching, damped to 0.4, for terms the corpus does not contain at + all. 'conditonal' is a typo, not a different question. + - Path queries. A query that looks like a CIPP route ('/identity/administration/users') + is matched against the page's own appPath, so an agent looking at a screen can ask for + that screen's documentation directly. + + The caller is itself a model and rephrases well, so the remaining gap to true semantic + recall is smaller in practice than it looks. Ranking stays behind this one function and + CIPP.DocsIndex.Search, which is where an embedding reranker would slot in if one ever + becomes available. + + Results are sections, not whole pages, and carry the heading anchor so the link lands on + the part that matched. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param( + [string]$Query, + # Restrict to a subtree, e.g. 'user-documentation/identity' or a CIPP route. + [string]$Path, + [int]$Limit = 8 + ) + + if ($Limit -lt 1) { $Limit = 8 } + if ($Limit -gt 25) { $Limit = 25 } + + $null = Get-CippDocsIndex + + if ([string]::IsNullOrWhiteSpace($Query) -and [string]::IsNullOrWhiteSpace($Path)) { + return [ordered]@{ + error = 'Provide a query (keywords or a question) and/or a path to scope the search.' + hint = 'Example: { "query": "how do I set up GDAP" } or { "path": "/identity/administration/users" }.' + } + } + + # A bare path with no keywords: return the pages under it directly. + if ([string]::IsNullOrWhiteSpace($Query)) { + $ByPath = [CIPP.DocsIndex]::ByPath($Path, $Limit) + if ($ByPath.MatchCount -eq 0) { + return [ordered]@{ + matchCount = 0 + results = @() + hint = "No documentation page matches path '$Path'. Search by keywords instead, or drop the path filter." + } + } + return [ordered]@{ + matchCount = $ByPath.MatchCount + results = @($ByPath.Hits | ForEach-Object { ConvertTo-CippDocSearchResult -Hit $_ }) + hint = 'Matched by path. Call GetDoc with a result''s path for the full page text.' + } + } + + $Primary = @(ConvertTo-CippDocToken -Text $Query) + if ($Primary.Count -eq 0) { + return [ordered]@{ + matchCount = 0 + results = @() + hint = 'The query reduced to no searchable terms. Try more specific keywords.' + } + } + + # Domain expansion happens here rather than in C# because the map is CIPP configuration, + # not an indexing concern; the index just takes the extra terms and damps them. + $Synonyms = Get-CippDocSynonym + $Expanded = [System.Collections.Generic.List[string]]::new() + foreach ($Token in $Primary) { + foreach ($Phrase in @($Synonyms[$Token])) { + if (-not $Phrase) { continue } + foreach ($Term in (ConvertTo-CippDocToken -Text $Phrase)) { + if ($Term -notin $Primary) { $Expanded.Add($Term) } + } + } + } + + $Search = [CIPP.DocsIndex]::Search([string[]]$Primary, [string[]]$Expanded.ToArray(), $Path, $Limit) + + if ($Search.MatchCount -eq 0) { + $Response = [ordered]@{ matchCount = 0; results = @() } + if ($Search.Suggestions.Count -gt 0) { + $Response['suggestions'] = @($Search.Suggestions) + $Response['hint'] = 'Nothing matched. The suggestions are the closest terms that do appear in the docs - try one of those.' + } else { + $Response['hint'] = 'Nothing matched. Try broader keywords, or the words a page would actually use.' + } + return $Response + } + + Write-Information "[MCP] SearchDocs query='$Query' path='$Path' -> $($Search.MatchCount) sections, returning $($Search.Hits.Count)" + + return [ordered]@{ + matchCount = $Search.MatchCount + results = @($Search.Hits | ForEach-Object { ConvertTo-CippDocSearchResult -Hit $_ }) + hint = 'Each result deep-links to the section that matched. Call GetDoc with a result''s path for the page''s full text.' + } +} + +function ConvertTo-CippDocSearchResult { + <# + .SYNOPSIS + Shapes a CIPPSharp search hit into the object returned to the MCP caller. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param([Parameter(Mandatory)]$Hit) + + $Result = [ordered]@{ + title = $Hit.Title + section = $Hit.Section + path = $Hit.Path + excerpt = $Hit.Excerpt + docsUrl = $Hit.DocsUrl + } + if (-not $Hit.Published) { + $Result['note'] = 'Not published on docs.cipp.app; use the GitHub link.' + } + $Result['githubUrl'] = $Hit.GitHubUrl + if ($Hit.AppPath) { $Result['appPath'] = $Hit.AppPath } + $Result['breadcrumb'] = $Hit.Breadcrumb + if ($Hit.Score -gt 0) { $Result['score'] = $Hit.Score } + + return $Result +} + +function Get-CippDocSynonym { + <# + .SYNOPSIS + Loads and caches the docs query-expansion map, keyed by stemmed token. + .DESCRIPTION + Config/DocsSynonyms.json is authored with ordinary words; the keys are stemmed here with + the same rule the indexer uses, so an author does not have to predict the stemmer. A + missing or malformed file degrades to no expansion rather than breaking search. + Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param([switch]$Force) + + if ($script:CippDocSynonym -and -not $Force) { return $script:CippDocSynonym } + + $Map = @{} + $Path = Join-Path -Path $env:CIPPRootPath -ChildPath 'Config/DocsSynonyms.json' + if (Test-Path -LiteralPath $Path) { + try { + $Json = [System.IO.File]::ReadAllText($Path) | ConvertFrom-Json -AsHashtable + foreach ($Pair in $Json.expansions.GetEnumerator()) { + $Map[[CIPP.DocsIndex]::Stem(([string]$Pair.Key).ToLowerInvariant())] = @($Pair.Value) + } + } catch { + Write-Information "[MCP] DocsSynonyms.json could not be read, continuing without query expansion: $($_.Exception.Message)" + } + } + + $script:CippDocSynonym = $Map + return $Map +} + +function Get-CippDoc { + <# + .SYNOPSIS + Returns the full text of one documentation page. + .DESCRIPTION + Backs the GetDoc core tool, for when a SearchDocs excerpt is not enough. Accepts whatever + identifier the caller happens to be holding - the repo-relative path from a search result, + the published slug, or the CIPP route the page documents - because an agent that found a + page one way should not have to convert it to another. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param([Parameter(Mandatory)][string]$Path) + + $null = Get-CippDocsIndex + + $Page = [CIPP.DocsIndex]::FindPage($Path) + if (-not $Page) { + # Deduplicated by page: a search returns up to two sections per page, and offering the + # same path twice as a 'did you mean' just wastes one of three suggestions. + $Near = @((Find-CippDoc -Query ($Path -replace '[/\-_.]', ' ') -Limit 6).results | + ForEach-Object { $_.path } | Select-Object -Unique | Select-Object -First 3) + return [ordered]@{ + error = "No documentation page matches '$Path'." + suggestions = @($Near) + hint = 'Find a page with SearchDocs first; its "path" is what this tool takes.' + } + } + + $Result = [ordered]@{ + title = $Page.Title + description = $Page.Description + path = $Page.RelativePath + breadcrumb = $Page.Breadcrumb + docsUrl = $Page.DocsUrl + githubUrl = $Page.GitHubUrl + } + if ($Page.AppPath) { $Result['appPath'] = $Page.AppPath } + if (-not $Page.Published) { + $Result['note'] = 'Not published on docs.cipp.app; use the GitHub link.' + } + $Result['sections'] = @($Page.Headings) + $Result['content'] = [CIPP.DocsIndex]::GetPageText($Page.RelativePath) + + return $Result +} diff --git a/backend/Modules/CIPPCore/Public/MCP/Get-CippDocLink.ps1 b/backend/Modules/CIPPCore/Public/MCP/Get-CippDocLink.ps1 new file mode 100644 index 0000000000..1848f726c1 --- /dev/null +++ b/backend/Modules/CIPPCore/Public/MCP/Get-CippDocLink.ps1 @@ -0,0 +1,125 @@ +function Get-CippDocLink { + <# + .SYNOPSIS + Derives the published docs.cipp.app URL, GitHub source URL and in-app route for a docs file. + .DESCRIPTION + The docs tree is a GitBook git-sync source, so a page's published URL is its file path with + the '.md' dropped - there is no slug table to consult. Three link forms come out of that: + + docs/setup/setting-up-cipp/install.md + -> https://docs.cipp.app/setup/setting-up-cipp/install + -> https://github.com/CyberDrain/CIPP/blob/dev/docs/setup/setting-up-cipp/install.md + + docs/setup/setting-up-cipp/README.md (a section index) + -> https://docs.cipp.app/setup/setting-up-cipp + + Pages under 'user-documentation/' mirror the frontend's own routing one-for-one, which is + the same assumption _app.js:259 already makes when it builds its "docs for this page" link. + That makes the mapping reversible: a doc knows which CIPP screen it documents, and a screen + can be traced back to its doc. Only paths under user-documentation get an AppPath; nothing + else in the tree corresponds to a route. + + Anchors are GitBook's heading slugs (lowercased, non-alphanumerics collapsed to hyphens), + which is what makes a chunk-level result deep-link to the exact section it matched rather + than to the top of a 25 KB page. + + One rule is not guessable from the path alone: a folder with no README.md is a grouping + folder, not a page, and GitBook drops it from the URL entirely. 'email/resources' has no + README and publishes nothing, so its children move up a level - + email/resources/management/equipment/edit.md is served at email/management/equipment/edit. + Without this, seven pages get confidently wrong links. -SectionFolder supplies the set of + folders that do own a README; the index builder computes it once for the whole tree. + + Only the docs URL is rewritten. The GitHub URL always keeps the real repo path, because + that is where the file actually lives. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param( + # Path of the markdown file relative to the docs root, e.g. 'setup/setting-up-cipp/install.md'. + [Parameter(Mandatory)] + [string]$RelativePath, + + # Optional heading text to deep-link to within the page. + [string]$Heading, + + # Folder paths (relative to the docs root, '/'-separated) that contain a README.md. + # Any ancestor folder absent from this set is elided from the published URL. + [System.Collections.Generic.HashSet[string]]$SectionFolder + ) + + $Clean = $RelativePath -replace '\\', '/' -replace '^\./', '' -replace '^/', '' + $Slug = $Clean -replace '(?i)\.md$', '' + + # A README is the index of the folder it sits in, so it publishes at the folder's own URL. + # The docs root README is the site root, which GitBook publishes as /readme rather than /. + if ($Slug -match '(?i)^README$') { + $Slug = 'readme' + } elseif ($Slug -match '(?i)/README$') { + $Slug = $Slug -replace '(?i)/README$', '' + } + + # Drop grouping folders. The final segment is the page itself and always survives. A + # top-level folder is a SUMMARY.md '## Group' and always contributes its slug even though + # it owns no README ('setup' has none, yet every setup page is served under /setup). + # Below that, a folder only earns a URL slot by owning a README. + if ($SectionFolder -and $Slug -ne 'readme') { + $Segments = @($Slug -split '/') + if ($Segments.Count -gt 1) { + $Kept = [System.Collections.Generic.List[string]]::new() + for ($i = 0; $i -lt $Segments.Count - 1; $i++) { + $Folder = ($Segments[0..$i] -join '/') + if ($i -eq 0 -or $SectionFolder.Contains($Folder)) { $Kept.Add($Segments[$i]) } + } + $Kept.Add($Segments[-1]) + $Slug = $Kept -join '/' + } + } + + $Anchor = if ($Heading) { Get-CippDocAnchor -Heading $Heading } else { '' } + $Fragment = if ($Anchor) { "#$Anchor" } else { '' } + + $AppPath = if ($Slug -match '^user-documentation/(.+)$') { "/$($Matches[1])" } else { $null } + + return [ordered]@{ + docsUrl = "https://docs.cipp.app/$Slug$Fragment" + githubUrl = "https://github.com/CyberDrain/CIPP/blob/dev/docs/$Clean$Fragment" + appPath = $AppPath + slug = $Slug + anchor = $Anchor + } +} + +function Get-CippDocAnchor { + <# + .SYNOPSIS + Converts a markdown heading to the anchor slug GitBook publishes for it. + .DESCRIPTION + Mirrors GitBook's slug rules: strip inline markdown, lowercase, drop anything that is not + alphanumeric or a space/hyphen, then collapse whitespace runs to single hyphens. Apostrophes + are removed rather than replaced, so "Confirm You've Met All Prerequisites" becomes + 'confirm-youve-met-all-prerequisites' and not 'confirm-you-ve-...'. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param([string]$Heading) + + if ([string]::IsNullOrWhiteSpace($Heading)) { return '' } + + $Text = $Heading -replace '^#+\s*', '' + # Inline markdown that renders away before the slug is taken: links keep their label only. + $Text = $Text -replace '\[([^\]]*)\]\([^)]*\)', '$1' + $Text = $Text -replace '[`*_~]', '' + # Straight and typographic apostrophes both vanish rather than becoming separators. + # Written as regex escapes, not literals: PowerShell reads a bare U+2019 as a quote + # delimiter, so a literal here is one encoding round-trip away from a parser error. + $Text = $Text -replace '[\u2018\u2019'']', '' + $Text = $Text.ToLowerInvariant() + $Text = $Text -replace '[^a-z0-9 \-]', ' ' + $Text = ($Text -replace '\s+', ' ').Trim() + $Text = $Text -replace '[\s\-]+', '-' + + return $Text.Trim('-') +} diff --git a/backend/Modules/CIPPCore/Public/MCP/Get-CippDocsIndex.ps1 b/backend/Modules/CIPPCore/Public/MCP/Get-CippDocsIndex.ps1 new file mode 100644 index 0000000000..8adb01b635 --- /dev/null +++ b/backend/Modules/CIPPCore/Public/MCP/Get-CippDocsIndex.ps1 @@ -0,0 +1,203 @@ +function Get-CippDocsIndex { + <# + .SYNOPSIS + Builds, once per host, the searchable index over the shipped CIPP documentation tree. + .DESCRIPTION + Backs the SearchDocs / GetDoc MCP tools. The docs are shipped in the image (see the docs/ + COPY in build/Dockerfile) rather than fetched from docs.cipp.app, because there is no + GitBook search API, llms-full.txt is capped at 100 of the 427 pages, and a crawl would put + an outbound-internet dependency in the request path. Shipping them also version-matches the + docs to the running build, and results still carry live docs.cipp.app links, so a caller + who needs the very latest text can always follow one. + + This function does discovery, markdown parsing and link derivation; CIPPSharp's + CIPP.DocsIndex owns tokenisation, the postings map and scoring. That split is deliberate: + the parsing is cheap and reads better in PowerShell, while tokenising 2 MB of prose in + PowerShell measured at 26 seconds against well under a second in .NET. Just as importantly + the C# index is a host-scoped static, so it is built once for every worker on the host + rather than once per runspace - the IsBuilt check below is what lets the other workers skip + all of this. + + Pages are split into chunks at '##'/'###' headings. A chunk, not a page, is the unit of + retrieval: pages here run to 25 KB, and returning a whole one to answer a question about a + single section wastes the caller's context and buries the answer. Each chunk carries its + heading anchor, so a hit deep-links to the exact section. + + Two page classes are excluded outright rather than ranked down: + - legacy-setup-hidden-from-nav/ 33 superseded 'Copy of ...' duplicates of the setup + guide. Near-identical text to the live pages, so they + would double every setup hit with a dead link. + - .gitbook/includes/ reusable snippets that are not pages at all. + Pages that exist but GitBook does not publish keep a GitHub link and get no docsUrl, + rather than being handed a docs.cipp.app URL that 404s. + + Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param( + # Overrides the docs root. Defaults to the shipped copy, then the repo tree for local dev. + [string]$DocsRoot, + [switch]$Force + ) + + if (-not $DocsRoot) { $DocsRoot = Get-CippDocsRoot } + if (-not $DocsRoot) { + throw [pscustomobject]@{ code = -32603; message = 'CIPP documentation not found in this deployment; docs search is unavailable.' } + } + + $DocsRoot = (Resolve-Path -LiteralPath $DocsRoot).Path.TrimEnd('\', '/') + + if (-not $Force -and [CIPP.DocsIndex]::IsBuilt($DocsRoot)) { + return Get-CippDocsIndexStatus -DocsRoot $DocsRoot + } + + $RootLength = $DocsRoot.Length + + # Folders owning a README are real pages and contribute a URL segment; the rest are elided. + $SectionFolder = [System.Collections.Generic.HashSet[string]]::new([StringComparer]::OrdinalIgnoreCase) + foreach ($Readme in [System.IO.Directory]::EnumerateFiles($DocsRoot, 'README.md', [System.IO.SearchOption]::AllDirectories)) { + $Dir = [System.IO.Path]::GetDirectoryName($Readme) + if ($Dir.Length -le $RootLength) { continue } + $SectionFolder.Add(($Dir.Substring($RootLength).TrimStart('\', '/') -replace '\\', '/')) | Out-Null + } + + # What GitBook actually publishes, so the index never invents a link. + $Published = Get-CippDocsPublishedSet + + $Builder = [CIPP.DocsIndex]::BeginBuild($DocsRoot) + + foreach ($FullPath in [System.IO.Directory]::EnumerateFiles($DocsRoot, '*.md', [System.IO.SearchOption]::AllDirectories)) { + $Rel = $FullPath.Substring($RootLength).TrimStart('\', '/') -replace '\\', '/' + + if ($Rel -eq 'SUMMARY.md') { continue } + if ($Rel -like '.gitbook/*') { continue } + if ($Rel -like 'legacy-setup-hidden-from-nav/*') { continue } + + $Parsed = ConvertFrom-CippDocMarkdown -Markdown ([System.IO.File]::ReadAllText($FullPath)) -RelativePath $Rel + $Link = Get-CippDocLink -RelativePath $Rel -SectionFolder $SectionFolder + + $IsPublished = if ($null -eq $Published) { $true } else { $Published.Contains($Link.slug) } + + $PageIndex = $Builder.AddPage( + $Rel, + $Parsed.Title, + $Parsed.Description, + $Parsed.Breadcrumb, + $Link.slug, + $(if ($IsPublished) { $Link.docsUrl } else { $null }), + $Link.githubUrl, + $(if ($IsPublished) { $Link.appPath } else { $null }), + $IsPublished) + + foreach ($Chunk in $Parsed.Chunks) { + if ([string]::IsNullOrWhiteSpace($Chunk.Text) -and -not $Chunk.Heading) { continue } + $Anchor = if ($Chunk.Heading) { Get-CippDocAnchor -Heading $Chunk.Heading } else { '' } + $Builder.AddChunk($PageIndex, $Chunk.Heading, $Anchor, $Chunk.Text) + } + } + + [CIPP.DocsIndex]::CommitBuild($Builder) + + $Status = Get-CippDocsIndexStatus -DocsRoot $DocsRoot + Write-Information "[MCP] docs index built: $($Status.pageCount) pages, $($Status.chunkCount) chunks, $($Status.termCount) terms from $DocsRoot" + return $Status +} + +function Get-CippDocsIndexStatus { + <# + .SYNOPSIS + Reports the current host-scoped docs index counts. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param([string]$DocsRoot) + + return [ordered]@{ + docsRoot = $DocsRoot + pageCount = [CIPP.DocsIndex]::PageCount + chunkCount = [CIPP.DocsIndex]::ChunkCount + termCount = [CIPP.DocsIndex]::TermCount + } +} + +function Get-CippDocsRoot { + <# + .SYNOPSIS + Locates the documentation tree, in the container or in a source checkout. + .DESCRIPTION + Checks CIPPDocsPath first (the dev compose files set it), then the image's own + $env:CIPPRootPath/Docs, then the repo layout so local dev and Pester runs work without a + build. Returns $null when none holds documentation. + + A candidate has to actually contain markdown to win, which is not the pedantry it looks + like. Bind-mounting the docs at /app/API/Docs - inside the ../backend mount - makes Docker + create the nested mountpoint on the *host*, leaving an empty backend/Docs in the working + tree. That directory then satisfies a bare existence check and shadows the real docs for + everything running outside the container, so every search silently returns nothing against + a perfectly healthy index of zero pages. The dev mount now lives at /app/Docs to avoid + creating it at all; this check is the backstop. Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param() + + $Candidates = [System.Collections.Generic.List[string]]::new() + if ($env:CIPPDocsPath) { $Candidates.Add($env:CIPPDocsPath) } + if ($env:CIPPRootPath) { + foreach ($Relative in 'Docs', '../docs', '../../docs') { + $Candidates.Add((Join-Path -Path $env:CIPPRootPath -ChildPath $Relative)) + } + } + + foreach ($Candidate in $Candidates) { + if (-not (Test-Path -LiteralPath $Candidate -PathType Container)) { continue } + # Select-Object -First 1 short-circuits the enumeration, so this stops at the first hit + # rather than walking the whole tree. + $Markdown = @([System.IO.Directory]::EnumerateFiles($Candidate, '*.md', [System.IO.SearchOption]::AllDirectories) | + Select-Object -First 1) + if ($Markdown.Count -gt 0) { return $Candidate } + } + return $null +} + +function Get-CippDocsPublishedSet { + <# + .SYNOPSIS + Reads the set of slugs GitBook actually publishes, from the committed snapshot. + .DESCRIPTION + A page can exist in docs/ and still not be live: nine current pages under + setup/implementation-guide/your-route-to-a-secure-tenant/ are in SUMMARY.md but + unpublished, so SUMMARY is not the discriminator. docs.cipp.app/llms.txt is the only + authoritative statement, and Config/DocsPublishedPages.txt is a snapshot of it - which + keeps the check offline, so the index never has to guess and never hands back a + docs.cipp.app URL that 404s. + + The container build refreshes that snapshot from the live site (the build-docspages stage), + so the committed copy is a fallback rather than the source of truth - it is what ships only + when the fetch fails. Refresh the committed one with + build/tools/Update-DocsPublishedPages.ps1. + + Returns $null when the snapshot is missing, which the caller reads as 'assume everything is + published' - a stale link is a better failure than no docs search at all. + Not an HTTP entrypoint. + .FUNCTIONALITY + Internal + #> + [CmdletBinding()] + param() + + $SnapshotPath = Join-Path -Path $env:CIPPRootPath -ChildPath 'Config/DocsPublishedPages.txt' + if (-not (Test-Path -LiteralPath $SnapshotPath)) { return $null } + + $Set = [System.Collections.Generic.HashSet[string]]::new([StringComparer]::OrdinalIgnoreCase) + foreach ($Line in [System.IO.File]::ReadAllLines($SnapshotPath)) { + $Trimmed = $Line.Trim() + if (-not $Trimmed -or $Trimmed.StartsWith('#')) { continue } + $Set.Add($Trimmed) | Out-Null + } + return $(if ($Set.Count -gt 0) { $Set } else { $null }) +} diff --git a/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolCatalog.ps1 b/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolCatalog.ps1 index c817a01917..329ade7d60 100644 --- a/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolCatalog.ps1 +++ b/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolCatalog.ps1 @@ -38,7 +38,11 @@ function Get-CippMcpToolCatalog { # backs the in-app documentation browser: it returns the whole ~1.5 MB OpenAPI # document, which would flood the caller's context to tell it what SearchTools # already answers. - if ($Endpoint -in @('ExecMcp', 'ListOpenApiSpec')) { continue } + # + # ListCippDocs is excluded for a different reason: it is already advertised as the + # SearchDocs and GetDoc core tools, and leaving it in the catalog would offer a + # third name for the same thing with a different argument shape. + if ($Endpoint -in @('ExecMcp', 'ListOpenApiSpec', 'ListCippDocs')) { continue } foreach ($MethodEntry in $PathEntry.Value.GetEnumerator()) { $Method = [string]$MethodEntry.Key diff --git a/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolList.ps1 b/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolList.ps1 index 46e6a4fa11..69d75ba30b 100644 --- a/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolList.ps1 +++ b/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolList.ps1 @@ -1,15 +1,24 @@ function Get-CippMcpToolList { <# .SYNOPSIS - Returns the fixed five-tool gateway advertised to MCP clients. + Returns the fixed core-tool gateway advertised to MCP clients. .DESCRIPTION tools/list never dumps the full read-only catalog (200+ tool schemas would flood the - client's context window). Instead every connection is offered the same five core tools: + client's context window). Instead every connection is offered the same core tools: - ListTenants direct passthrough; the entry point for tenant resolution - ListGraphRequest direct passthrough; arbitrary Microsoft Graph GET proxy - SearchTools browse/search the catalog (compact results, no schemas) - GetToolInfo full description + inputSchema for named catalog tools - ExecTool execute any catalog tool by name + - SearchDocs search the CIPP documentation + - GetDoc fetch one documentation page in full + + SearchDocs and GetDoc are advertised rather than left to be discovered through + SearchTools because they answer a different kind of question from the rest of the + catalog. Every other tool returns tenant data; these explain what CIPP does and how to + drive it, which is exactly what an agent needs *before* it knows which data tool to + reach for. A model that has to already suspect the docs exist in order to find them + will simply guess at CIPP's behaviour instead. The connector URL's query filters (?tags=, ?tools=, ?first=) scope the catalog visible to SearchTools/GetToolInfo/ExecTool; the five core tools themselves are always advertised. Passthrough schemas are projected live from the catalog so they track openapi.json. @@ -82,6 +91,33 @@ function Get-CippMcpToolList { annotations = [ordered]@{ title = 'ExecTool'; readOnlyHint = $true } }) + $Tools.Add([ordered]@{ + name = 'SearchDocs' + description = 'Search the CIPP documentation (docs.cipp.app) by keywords or a plain-language question. Use this to find out how a CIPP feature works, how to configure something, or what a screen does - it answers "how do I" and "what is", where the other tools return tenant data. Results are individual sections with an excerpt and a link that deep-links to the matching heading. Pass path to scope to one area, or to look up the documentation for a CIPP screen by its route (e.g. "/identity/administration/users").' + inputSchema = [ordered]@{ + type = 'object' + properties = [ordered]@{ + query = @{ type = 'string'; description = 'Keywords or a question, e.g. "how do I set up GDAP" or "conditional access templates".' } + path = @{ type = 'string'; description = 'Optional. Restrict to a documentation subtree ("user-documentation/identity") or a CIPP route ("/identity/administration/users").' } + limit = @{ type = 'integer'; description = 'Maximum results (default 8, max 25).' } + } + } + annotations = [ordered]@{ title = 'SearchDocs'; readOnlyHint = $true } + }) + + $Tools.Add([ordered]@{ + name = 'GetDoc' + description = 'Fetch one CIPP documentation page in full, when a SearchDocs excerpt is not enough. Takes the "path" from a SearchDocs result, a published docs.cipp.app slug, or the CIPP route the page documents.' + inputSchema = [ordered]@{ + type = 'object' + properties = [ordered]@{ + path = @{ type = 'string'; description = 'The page to fetch, as returned in a SearchDocs result''s "path".' } + } + required = @('path') + } + annotations = [ordered]@{ title = 'GetDoc'; readOnlyHint = $true } + }) + Write-Information "[MCP] tools/list -> $($Tools.Count) core tools (catalog=$FilteredCount)" return @($Tools) diff --git a/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolResult.ps1 b/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolResult.ps1 index 4190c5b9c6..e52e26ef01 100644 --- a/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolResult.ps1 +++ b/backend/Modules/CIPPCore/Public/MCP/Get-CippMcpToolResult.ps1 @@ -3,7 +3,8 @@ function Get-CippMcpToolResult { .SYNOPSIS Dispatches a single MCP 'tools/call' through the five-tool gateway. .DESCRIPTION - SearchTools and GetToolInfo are answered locally from the read-only tool catalog. + SearchTools and GetToolInfo are answered locally from the read-only tool catalog, and + SearchDocs / GetDoc from the documentation index shipped in the image. ListTenants, ListGraphRequest and ExecTool targets are re-dispatched through New-CippCoreRequest via Invoke-CippMcpApiRequest, so RBAC and tenant scoping are enforced on every execution. Direct calls to bare catalog tool names are still accepted for @@ -77,6 +78,29 @@ function Get-CippMcpToolResult { isError = $false } } + 'SearchDocs' { + # Answered locally, like SearchTools: the documentation is shipped in the image and + # is not tenant data, so there is nothing to re-dispatch through the API and no + # tenant scoping to enforce. + $Limit = $ArgHash['limit'] -as [int] + if (-not $Limit) { $Limit = 8 } + $DocResult = Find-CippDoc -Query ([string]$ArgHash['query']) -Path ([string]$ArgHash['path']) -Limit $Limit + return [ordered]@{ + content = @(@{ type = 'text'; text = ($DocResult | ConvertTo-Json -Depth 10 -Compress) }) + isError = [bool]$DocResult.error + } + } + 'GetDoc' { + $DocPath = [string]($ArgHash['path'] ?? $ArgHash['name']) + if ([string]::IsNullOrWhiteSpace($DocPath)) { + throw [pscustomobject]@{ code = -32602; message = 'Invalid params: path (from a SearchDocs result) is required' } + } + $Doc = Get-CippDoc -Path $DocPath + return [ordered]@{ + content = @(@{ type = 'text'; text = ($Doc | ConvertTo-Json -Depth 10 -Compress) }) + isError = [bool]$Doc.error + } + } 'ExecTool' { $TargetName = [string]$ArgHash['name'] if ([string]::IsNullOrWhiteSpace($TargetName)) { diff --git a/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/Core/Invoke-ListCippDocs.ps1 b/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/Core/Invoke-ListCippDocs.ps1 new file mode 100644 index 0000000000..298a59d633 --- /dev/null +++ b/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/Core/Invoke-ListCippDocs.ps1 @@ -0,0 +1,52 @@ +function Invoke-ListCippDocs { + <# + .FUNCTIONALITY + Entrypoint,AnyTenant + .ROLE + CIPP.Core.Read + .SYNOPSIS + Search the CIPP documentation, or fetch one documentation page in full. + .DESCRIPTION + Searches the GitBook documentation shipped with this build and returns matching sections, + each with an excerpt and links back to docs.cipp.app and to the file on GitHub. Pages under + user-documentation also report the CIPP route they document, so a screen can be traced to + its docs and back. + + Pass path on its own to list the pages under a documentation subtree or a CIPP route, or + with full=true to return one page's entire text. This backs the SearchDocs and GetDoc MCP + tools and is available to the UI and API clients on the same terms. + #> + [CmdletBinding()] + param($Request, $TriggerMetadata) + + $Headers = $Request.Headers + + # Keywords or a plain-language question, e.g. 'how do I set up GDAP'. + $Query = $Request.Query.query ?? $Request.Body.query + # A documentation subtree ('user-documentation/identity') or a CIPP route + # ('/identity/administration/users'). + $Path = $Request.Query.path ?? $Request.Body.path + # Return the whole page rather than matching sections. Requires path. + $Full = [bool]($Request.Query.full ?? $Request.Body.full) + # Maximum results to return (default 8, max 25). + $Limit = ($Request.Query.limit ?? $Request.Body.limit) -as [int] + + try { + if ($Full) { + if (-not $Path) { throw 'path is required when full=true.' } + $Result = Get-CippDoc -Path $Path + } else { + $Result = Find-CippDoc -Query ([string]$Query) -Path ([string]$Path) -Limit $Limit + } + $StatusCode = [HttpStatusCode]::OK + } catch { + Write-LogMessage -API 'ListCippDocs' -message "Documentation search failed: $($_.Exception.Message)" -sev Error -headers $Headers + $StatusCode = [HttpStatusCode]::InternalServerError + $Result = @{ error = $_.Exception.Message } + } + + return ([HttpResponseContext]@{ + StatusCode = $StatusCode + Body = $Result + }) +} diff --git a/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/MCP/Invoke-ExecMcp.ps1 b/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/MCP/Invoke-ExecMcp.ps1 index 3028c8616c..f27909e8ff 100644 --- a/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/MCP/Invoke-ExecMcp.ps1 +++ b/backend/Modules/CIPPHTTP/Public/Entrypoints/HTTP Functions/CIPP/MCP/Invoke-ExecMcp.ps1 @@ -82,7 +82,7 @@ function Invoke-ExecMcp { name = 'CIPP' version = $Request.Headers.'X-CIPP-Version' ?? 'unknown' } - instructions = 'CIPP is a gateway to the read-only CIPP API. Five tools are exposed: ListTenants (enumerate managed tenants; most tools need a tenantFilter — use the tenant''s defaultDomainName), ListGraphRequest (proxy an arbitrary Microsoft Graph GET), SearchTools (browse or keyword-search the full tool catalog), GetToolInfo (fetch a tool''s input schema), and ExecTool (run any discovered tool by name). Typical flow: ListTenants -> SearchTools -> GetToolInfo -> ExecTool.' + instructions = 'CIPP is a gateway to the read-only CIPP API. Seven tools are exposed: ListTenants (enumerate managed tenants; most tools need a tenantFilter — use the tenant''s defaultDomainName), ListGraphRequest (proxy an arbitrary Microsoft Graph GET), SearchTools (browse or keyword-search the full tool catalog), GetToolInfo (fetch a tool''s input schema), ExecTool (run any discovered tool by name), SearchDocs (search the CIPP documentation) and GetDoc (fetch one documentation page in full). Typical flow for data: ListTenants -> SearchTools -> GetToolInfo -> ExecTool. For questions about how CIPP works or how to configure it, start with SearchDocs rather than guessing.' } } 'ping' { $Result = @{} } diff --git a/backend/Shared/CIPPSharp/CippDocsIndex.cs b/backend/Shared/CIPPSharp/CippDocsIndex.cs new file mode 100644 index 0000000000..0e177c9fb1 --- /dev/null +++ b/backend/Shared/CIPPSharp/CippDocsIndex.cs @@ -0,0 +1,673 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text; + +namespace CIPP +{ + /// + /// Host-scoped inverted index over the shipped CIPP documentation, backing the SearchDocs + /// and GetDoc MCP tools. + /// + /// This lives in C# for two reasons, both measured rather than assumed. Tokenising the + /// 2 MB corpus in PowerShell took 26 seconds, because the inner loop runs some 400k times + /// and PowerShell pays interpreter and call overhead on every iteration; the same work here + /// is well under a second. More importantly the index is static, so - exactly as with + /// TestDataCache - the DLL is loaded once per host and every PowerShell worker on that host + /// shares this one instance. A PowerShell $script: cache is per-runspace, so a worker pool + /// would have rebuilt the whole index once per worker, and warming it on a timer would only + /// ever have warmed the single runspace the timer happened to run in. + /// + /// The caller supplies pages and chunks already parsed and linked (that logic stays in + /// PowerShell, where it is cheap and readable); this class owns tokenisation, the postings + /// map, BM25 scoring, fuzzy vocabulary matching and excerpting. + /// + public static class DocsIndex + { + // ── BM25 parameters ── + private const double K1 = 1.2; + private const double B = 0.75; + + // Damping for expanded terms. A synonym or a typo-correction can promote a page but must + // never outrank a chunk that matched what the caller actually typed. + private const double SynonymWeight = 0.55; + private const double FuzzyWeight = 0.40; + + // ── State ── + private static readonly object _buildLock = new(); + private static volatile IndexData? _current; + + public sealed class PageRecord + { + public string RelativePath = ""; + public string Title = ""; + public string Description = ""; + public string Breadcrumb = ""; + public string Slug = ""; + public string? DocsUrl; + public string GitHubUrl = ""; + public string? AppPath; + public bool Published; + public List Headings = new(); + } + + internal sealed class ChunkRecord + { + public int PageIndex; + public string Heading = ""; + public string Anchor = ""; + public string Text = ""; + public int Length; + } + + internal sealed class IndexData + { + public string Key = ""; + public List Pages = new(); + public List Chunks = new(); + public Dictionary> Postings = + new(StringComparer.Ordinal); + public double AverageLength = 1; + } + + /// + /// Accumulates an index. Handed back to the caller as an instance rather than kept in a + /// static: a half-built index must never be reachable from another worker, and an + /// abandoned build (an exception mid-loop) has to collect rather than wedge the host. + /// + public sealed class Builder + { + internal readonly IndexData Data; + internal Builder(string key) { Data = new IndexData { Key = key }; } + + public int PageCount => Data.Pages.Count; + public int ChunkCount => Data.Chunks.Count; + + public int AddPage(string relativePath, string title, string description, + string breadcrumb, string slug, string? docsUrl, string gitHubUrl, string? appPath, + bool published) + { + Data.Pages.Add(new PageRecord + { + RelativePath = relativePath ?? "", + Title = title ?? "", + Description = description ?? "", + Breadcrumb = breadcrumb ?? "", + Slug = slug ?? "", + DocsUrl = string.IsNullOrWhiteSpace(docsUrl) ? null : docsUrl, + GitHubUrl = gitHubUrl ?? "", + AppPath = string.IsNullOrWhiteSpace(appPath) ? null : appPath, + Published = published + }); + return Data.Pages.Count - 1; + } + + /// + /// Adds one heading-delimited chunk. The page's title, description and slug words are + /// folded in at a boost weight rather than being repeated into the text by the caller: + /// a 20-section page would otherwise have its title tokenised 20 times over. + /// + public void AddChunk(int pageIndex, string? heading, string? anchor, string? text) + { + if (pageIndex < 0 || pageIndex >= Data.Pages.Count) return; + + var page = Data.Pages[pageIndex]; + heading ??= ""; + text ??= ""; + + var frequency = new Dictionary(StringComparer.Ordinal); + int length = 0; + + // Weights: what a chunk is *about* outranks a word it merely contains. + length += Accumulate(frequency, page.Title, 3); + length += Accumulate(frequency, heading, 3); + length += Accumulate(frequency, page.Description, 2); + length += Accumulate(frequency, page.Slug.Replace('/', ' ').Replace('-', ' '), 1); + length += Accumulate(frequency, text, 1); + + int chunkId = Data.Chunks.Count; + foreach (var pair in frequency) + { + if (!Data.Postings.TryGetValue(pair.Key, out var list)) + { + list = new List<(int, int)>(); + Data.Postings[pair.Key] = list; + } + list.Add((chunkId, pair.Value)); + } + + if (!string.IsNullOrEmpty(heading)) page.Headings.Add(heading); + + Data.Chunks.Add(new ChunkRecord + { + PageIndex = pageIndex, + Heading = heading, + Anchor = anchor ?? "", + Text = text, + Length = length + }); + } + + private static int Accumulate(Dictionary frequency, string text, int weight) + { + int added = 0; + foreach (var token in Tokenize(text)) + { + frequency.TryGetValue(token, out int current); + frequency[token] = current + weight; + added += weight; + } + return added; + } + } + + /// + /// A search's hits plus the total that matched before the per-page cap and limit. + /// Returned rather than exposed as a static counter: this class is shared by every + /// worker on the host, so a mutable static would be clobbered by concurrent searches. + /// + public sealed class SearchResult + { + public int MatchCount; + public SearchHit[] Hits = Array.Empty(); + public string[] Suggestions = Array.Empty(); + } + + /// Result row handed back to PowerShell, already shaped for the MCP response. + public sealed class SearchHit + { + public string Title = ""; + public string? Section; + public string Path = ""; + public string Excerpt = ""; + public string? DocsUrl; + public string GitHubUrl = ""; + public string? AppPath; + public string Breadcrumb = ""; + public bool Published; + public double Score; + } + + // ────────────────────────────────── Build ────────────────────────────────── + + /// True when an index for this key is already available on this host. + public static bool IsBuilt(string key) + { + var current = _current; + return current != null && string.Equals(current.Key, key, StringComparison.OrdinalIgnoreCase); + } + + /// Starts a new index build. The returned builder is the caller's to hold. + public static Builder BeginBuild(string key) => new Builder(key); + + /// + /// Publishes a completed index. The swap is a single reference assignment to a volatile + /// field, so a concurrent reader sees either the previous index or the new one, never a + /// half-populated one. Two workers racing to build simply duplicate the work and the last + /// to finish wins - both produce the same index. + /// + public static void CommitBuild(Builder builder) + { + if (builder == null) throw new ArgumentNullException(nameof(builder)); + + builder.Data.AverageLength = builder.Data.Chunks.Count > 0 + ? builder.Data.Chunks.Average(c => (double)c.Length) + : 1; + + lock (_buildLock) { _current = builder.Data; } + } + + public static void Clear() + { + lock (_buildLock) { _current = null; } + } + + public static int PageCount => _current?.Pages.Count ?? 0; + public static int ChunkCount => _current?.Chunks.Count ?? 0; + public static int TermCount => _current?.Postings.Count ?? 0; + + // ──────────────────────────────── Tokenising ──────────────────────────────── + + private static readonly HashSet StopWords = new(StringComparer.Ordinal) + { + "the","and","for","are","but","wa","were","been","being","have","ha","had", + "that","thi","these","those","with","from","into","onto","your","you","their", + "them","they","it","be","is","of","to","in","on","at","by","or","as", + "an","if","then","than","so","such","via","per","each","any","more","most", + "other","some","will","can","may","must","should","would","could","when","where", + "which","who","what","how","why","here","there","also","about","over","under", + "after","before","between","both","only","own","same","too","very","just","do", + "doe","did","done","get","got","make","made","use","used","using","want" + }; + + /// + /// The single tokenisation rule, shared by indexing and querying - the two must agree + /// exactly or a term indexed as 'standard' is never found by a query for 'Standards'. + /// + /// Compound identifiers are kept whole *and* split, because both spellings get searched: + /// 'ListUsers' yields listuser + list + user, 'Identity.User.ReadWrite' the whole string + /// plus its parts. Without the split a search for 'user permissions' misses a page that + /// only writes the role name; without the whole form, an exact search for the role name + /// ranks no better than one for 'user'. + /// + public static string[] Tokenize(string? text) + { + if (string.IsNullOrWhiteSpace(text)) return Array.Empty(); + + var tokens = new List(); + int i = 0; + int length = text!.Length; + + while (i < length) + { + // Scan one raw word: letters, digits, and the identifier punctuation we keep. + while (i < length && !IsWordChar(text[i])) i++; + int start = i; + while (i < length && IsWordChar(text[i])) i++; + if (i == start) continue; + + var raw = text.Substring(start, i - start).Trim('.', '-', '_'); + if (raw.Length < 2) continue; + + bool compound = false; + bool hasLowerUpper = false; + for (int k = 0; k < raw.Length; k++) + { + char c = raw[k]; + if (c == '.' || c == '-' || c == '_') compound = true; + if (k > 0 && char.IsUpper(c) && char.IsLower(raw[k - 1])) hasLowerUpper = true; + } + + var whole = raw.ToLowerInvariant(); + AddStemmed(tokens, whole); + + if (!compound && !hasLowerUpper) continue; + + // Split camelCase, then on the identifier punctuation. + var spaced = new StringBuilder(raw.Length + 8); + for (int k = 0; k < raw.Length; k++) + { + char c = raw[k]; + if (k > 0 && char.IsUpper(c) && (char.IsLower(raw[k - 1]) || char.IsDigit(raw[k - 1]))) + spaced.Append(' '); + spaced.Append(c == '.' || c == '-' || c == '_' ? ' ' : c); + } + + foreach (var part in spaced.ToString().Split(' ', StringSplitOptions.RemoveEmptyEntries)) + { + if (part.Length < 2) continue; + var lower = part.ToLowerInvariant(); + if (lower == whole) continue; + AddStemmed(tokens, lower); + } + } + + return tokens.ToArray(); + } + + private static bool IsWordChar(char c) => + char.IsLetterOrDigit(c) || c == '.' || c == '-' || c == '_'; + + private static void AddStemmed(List tokens, string lower) + { + var stem = Stem(lower); + if (!StopWords.Contains(stem)) tokens.Add(stem); + } + + /// + /// Light stemming - plural 's'/'es' only. An aggressive stemmer conflates CIPP vocabulary + /// that has to stay distinct, and the corpus is small enough that recall is not the + /// problem the stemmer would be solving. + /// + public static string Stem(string token) + { + // Too short to suffix-strip without destroying the word ('ies' -> '', 'use' -> 'us'). + if (token.Length <= 4) return token; + if (token.EndsWith("ies", StringComparison.Ordinal)) + return string.Concat(token.AsSpan(0, token.Length - 3), "y"); + if (token.EndsWith("sses", StringComparison.Ordinal) || token.EndsWith("shes", StringComparison.Ordinal) + || token.EndsWith("ches", StringComparison.Ordinal) || token.EndsWith("xes", StringComparison.Ordinal)) + return token.Substring(0, token.Length - 2); + if (token.EndsWith("ss", StringComparison.Ordinal)) return token; + if (token.EndsWith("s", StringComparison.Ordinal)) return token.Substring(0, token.Length - 1); + return token; + } + + // ────────────────────────────────── Search ────────────────────────────────── + + /// + /// Ranks chunks for a query. come from the caller's + /// domain expansion map; fuzzy correction is applied here, only for primary terms the + /// corpus does not contain at all - a term that matched exactly needs no help, and + /// fuzzing it would drag in neighbours that dilute a perfectly good query. + /// + public static SearchResult Search(string[]? primaryTerms, string[]? synonymTerms, + string? pathFilter, int limit, int perPageCap = 2) + { + var empty = new SearchResult(); + var data = _current; + if (data == null || data.Chunks.Count == 0) return empty; + if (limit < 1) limit = 8; + + primaryTerms ??= Array.Empty(); + synonymTerms ??= Array.Empty(); + + var allowedPages = ResolvePathFilter(data, pathFilter); + if (allowedPages != null && allowedPages.Count == 0) return empty; + + var weighted = new Dictionary(StringComparer.Ordinal); + foreach (var term in primaryTerms) + if (!string.IsNullOrEmpty(term)) weighted[term] = 1.0; + + foreach (var term in synonymTerms) + if (!string.IsNullOrEmpty(term) && !weighted.ContainsKey(term)) + weighted[term] = SynonymWeight; + + foreach (var term in primaryTerms) + { + if (string.IsNullOrEmpty(term) || data.Postings.ContainsKey(term)) continue; + foreach (var near in FuzzyMatches(data, term, 3)) + if (!weighted.ContainsKey(near)) weighted[near] = FuzzyWeight; + } + + if (weighted.Count == 0) return empty; + + var scores = new Dictionary(); + var covered = new Dictionary>(); + double chunkCount = data.Chunks.Count; + + foreach (var (term, weight) in weighted) + { + if (!data.Postings.TryGetValue(term, out var postings)) continue; + + double documentFrequency = postings.Count; + double idf = Math.Log(1 + (chunkCount - documentFrequency + 0.5) / (documentFrequency + 0.5)); + + foreach (var (chunkId, frequency) in postings) + { + if (allowedPages != null && !allowedPages.Contains(data.Chunks[chunkId].PageIndex)) continue; + + double len = data.Chunks[chunkId].Length; + double denominator = frequency + K1 * (1 - B + B * (len / Math.Max(data.AverageLength, 1))); + double contribution = weight * idf * (frequency * (K1 + 1) / Math.Max(denominator, 0.0001)); + + scores.TryGetValue(chunkId, out double running); + scores[chunkId] = running + contribution; + + if (!covered.TryGetValue(chunkId, out var set)) + { + set = new HashSet(StringComparer.Ordinal); + covered[chunkId] = set; + } + set.Add(term); + } + } + + if (scores.Count == 0) + { + return new SearchResult { Suggestions = Suggest(primaryTerms) }; + } + + // Coverage bonus: a chunk hitting three of the query's terms is answering the whole + // question; one hitting the same term three times is a page that just says it a lot. + double primaryCount = Math.Max(primaryTerms.Length, 1); + foreach (var chunkId in scores.Keys.ToArray()) + { + double coverage = covered[chunkId].Count / primaryCount; + scores[chunkId] *= 1 + 0.35 * Math.Min(coverage, 1.5); + } + + var hits = new List(); + var takenPerPage = new Dictionary(); + + foreach (var entry in scores.OrderByDescending(e => e.Value)) + { + if (hits.Count >= limit) break; + var chunk = data.Chunks[entry.Key]; + + // One page should not occupy the whole result set with five of its own sections. + takenPerPage.TryGetValue(chunk.PageIndex, out int taken); + if (taken >= perPageCap) continue; + takenPerPage[chunk.PageIndex] = taken + 1; + + hits.Add(BuildHit(data.Pages[chunk.PageIndex], chunk, Math.Round(entry.Value, 3), + primaryTerms.Concat(synonymTerms).ToArray())); + } + + return new SearchResult { MatchCount = scores.Count, Hits = hits.ToArray() }; + } + + private static HashSet? ResolvePathFilter(IndexData data, string? pathFilter) + { + if (string.IsNullOrWhiteSpace(pathFilter)) return null; + + var needle = pathFilter!.Replace('\\', '/').Trim().Trim('/').ToLowerInvariant(); + var allowed = new HashSet(); + + for (int i = 0; i < data.Pages.Count; i++) + { + var page = data.Pages[i]; + var slug = page.Slug.ToLowerInvariant(); + var appPath = (page.AppPath ?? "").Trim('/').ToLowerInvariant(); + + if (slug == needle || slug.StartsWith(needle + "/", StringComparison.Ordinal) + || slug.EndsWith("/" + needle, StringComparison.Ordinal) + || (appPath.Length > 0 && (appPath == needle + || appPath.StartsWith(needle + "/", StringComparison.Ordinal)))) + { + allowed.Add(i); + } + } + return allowed; + } + + /// Pages matching a path, for a path-only query with no keywords. + public static SearchResult ByPath(string pathFilter, int limit) + { + var data = _current; + if (data == null) return new SearchResult(); + + var allowed = ResolvePathFilter(data, pathFilter); + if (allowed == null || allowed.Count == 0) return new SearchResult(); + + var hits = new List(); + // Shortest slug first: the section index is a better first answer than a leaf page. + foreach (var pageIndex in allowed.OrderBy(p => data.Pages[p].Slug.Length).Take(limit)) + { + var intro = data.Chunks.FirstOrDefault(c => c.PageIndex == pageIndex) + ?? new ChunkRecord { PageIndex = pageIndex }; + hits.Add(BuildHit(data.Pages[pageIndex], intro, 0, Array.Empty())); + } + return new SearchResult { MatchCount = allowed.Count, Hits = hits.ToArray() }; + } + + private static SearchHit BuildHit(PageRecord page, ChunkRecord chunk, double score, string[] terms) + { + var fragment = string.IsNullOrEmpty(chunk.Anchor) ? "" : "#" + chunk.Anchor; + return new SearchHit + { + Title = page.Title, + Section = string.IsNullOrEmpty(chunk.Heading) ? null : chunk.Heading, + Path = page.RelativePath, + Excerpt = Excerpt(chunk.Text, terms), + DocsUrl = page.DocsUrl == null ? null : page.DocsUrl + fragment, + GitHubUrl = page.GitHubUrl + fragment, + AppPath = page.AppPath, + Breadcrumb = page.Breadcrumb, + Published = page.Published, + Score = score + }; + } + + // ─────────────────────────────── Fuzzy matching ─────────────────────────────── + + private static List FuzzyMatches(IndexData data, string token, int maxResults) + { + var results = new List<(string Term, double Distance, int Frequency)>(); + if (token.Length < 4) return new List(); + + int budget = token.Length >= 7 ? 2 : 1; + char first = token[0]; + + foreach (var candidate in data.Postings.Keys) + { + if (Math.Abs(candidate.Length - token.Length) > budget) + { + // A longer vocabulary term the query prefixes is still a good lead: + // 'conditional' should reach 'conditionalaccess'. + if (candidate.Length > token.Length && candidate.StartsWith(token, StringComparison.Ordinal)) + results.Add((candidate, 0.5, data.Postings[candidate].Count)); + continue; + } + if (candidate.Length == 0 || candidate[0] != first) continue; + + int distance = EditDistance(token, candidate, budget); + if (distance <= budget) results.Add((candidate, distance, data.Postings[candidate].Count)); + } + + return results + .OrderBy(r => r.Distance) + .ThenByDescending(r => r.Frequency) + .Take(maxResults) + .Select(r => r.Term) + .ToList(); + } + + /// Levenshtein distance, abandoning the row once it exceeds the ceiling. + public static int EditDistance(string first, string second, int ceiling = 2) + { + if (first.Length == 0) return second.Length; + if (second.Length == 0) return first.Length; + + var previous = new int[second.Length + 1]; + var current = new int[second.Length + 1]; + for (int j = 0; j <= second.Length; j++) previous[j] = j; + + for (int i = 1; i <= first.Length; i++) + { + current[0] = i; + int rowMinimum = current[0]; + for (int j = 1; j <= second.Length; j++) + { + int cost = first[i - 1] == second[j - 1] ? 0 : 1; + current[j] = Math.Min(Math.Min(current[j - 1] + 1, previous[j] + 1), previous[j - 1] + cost); + if (current[j] < rowMinimum) rowMinimum = current[j]; + } + if (rowMinimum > ceiling) return ceiling + 1; + (previous, current) = (current, previous); + } + return previous[second.Length]; + } + + /// Closest real vocabulary terms, for the "did you mean" path on a zero-result query. + public static string[] Suggest(string[] terms, int maxResults = 5) + { + var data = _current; + if (data == null) return Array.Empty(); + + var suggestions = new List(); + foreach (var term in terms) + { + foreach (var near in FuzzyMatches(data, term, 2)) + { + if (suggestions.Count >= maxResults) break; + if (!suggestions.Contains(near)) suggestions.Add(near); + } + } + return suggestions.ToArray(); + } + + // ─────────────────────────────────── Pages ─────────────────────────────────── + + /// Fetches a page by repo-relative path, slug or app route, for GetDoc. + public static PageRecord? FindPage(string pathOrSlug) + { + var data = _current; + if (data == null || string.IsNullOrWhiteSpace(pathOrSlug)) return null; + + var needle = pathOrSlug.Replace('\\', '/').Trim().Trim('/').ToLowerInvariant(); + var withoutExtension = needle.EndsWith(".md", StringComparison.Ordinal) + ? needle.Substring(0, needle.Length - 3) : needle; + + PageRecord? bySlug = null, byApp = null, bySuffix = null; + foreach (var page in data.Pages) + { + var relative = page.RelativePath.ToLowerInvariant(); + if (relative == needle || relative == needle + ".md") return page; + + var slug = page.Slug.ToLowerInvariant(); + if (slug == withoutExtension) bySlug ??= page; + if ((page.AppPath ?? "").Trim('/').ToLowerInvariant() == withoutExtension) byApp ??= page; + if (relative.EndsWith("/" + withoutExtension + ".md", StringComparison.Ordinal)) bySuffix ??= page; + } + return bySlug ?? byApp ?? bySuffix; + } + + /// The full text of a page, reassembled from its chunks with headings restored. + public static string GetPageText(string relativePath) + { + var data = _current; + if (data == null) return ""; + + var builder = new StringBuilder(); + for (int i = 0; i < data.Pages.Count; i++) + { + if (!string.Equals(data.Pages[i].RelativePath, relativePath, StringComparison.OrdinalIgnoreCase)) + continue; + + builder.Append("# ").AppendLine(data.Pages[i].Title); + foreach (var chunk in data.Chunks.Where(c => c.PageIndex == i)) + { + if (!string.IsNullOrEmpty(chunk.Heading)) + builder.AppendLine().Append("## ").AppendLine(chunk.Heading); + if (!string.IsNullOrEmpty(chunk.Text)) builder.AppendLine(chunk.Text); + } + break; + } + return builder.ToString().Trim(); + } + + public static PageRecord[] GetPages() => _current?.Pages.ToArray() ?? Array.Empty(); + + // ────────────────────────────────── Excerpts ────────────────────────────────── + + /// + /// A readable window of the chunk centred on the query's terms, so the caller can judge + /// relevance without a second call. Falls back to the opening sentence, which is a fair + /// summary of a section. + /// + private static string Excerpt(string text, string[] terms, int width = 320) + { + if (string.IsNullOrWhiteSpace(text)) return ""; + + var flat = System.Text.RegularExpressions.Regex.Replace(text, @"\s+", " ").Trim(); + if (flat.Length <= width) return flat; + + int best = 0; + if (terms.Length > 0) + { + var lower = flat.ToLowerInvariant(); + int bestHits = -1; + // Sampled window starts, not every offset: the excerpt only needs to be + // representative, and this keeps a 25 KB section cheap to summarise. + for (int start = 0; start < flat.Length - 1; start += 40) + { + int span = Math.Min(width, lower.Length - start); + var slice = lower.Substring(start, span); + int hits = terms.Count(t => t.Length > 0 && slice.Contains(t, StringComparison.Ordinal)); + if (hits > bestHits) { bestHits = hits; best = start; } + } + if (bestHits <= 0) best = 0; + } + + if (best > 0) + { + int space = flat.LastIndexOf(' ', Math.Min(best, flat.Length - 1)); + if (space > 0) best = space + 1; + } + + var excerpt = flat.Substring(best, Math.Min(width, flat.Length - best)).Trim(); + return (best > 0 ? "..." : "") + excerpt + (best + width < flat.Length ? "..." : ""); + } + } +} diff --git a/backend/Shared/CIPPSharp/bin/CIPPSharp.dll b/backend/Shared/CIPPSharp/bin/CIPPSharp.dll index ba68cf80f0..f66102fb69 100644 Binary files a/backend/Shared/CIPPSharp/bin/CIPPSharp.dll and b/backend/Shared/CIPPSharp/bin/CIPPSharp.dll differ diff --git a/backend/Tests/Mcp/CippDocs.Tests.ps1 b/backend/Tests/Mcp/CippDocs.Tests.ps1 new file mode 100644 index 0000000000..850f5385b5 --- /dev/null +++ b/backend/Tests/Mcp/CippDocs.Tests.ps1 @@ -0,0 +1,429 @@ +# Pester tests for the docs search tools (SearchDocs / GetDoc). +# +# The load-bearing claim these tools make is that every result links somewhere real. A search +# result is only useful if its docs.cipp.app URL resolves, and the URL is derived from the file's +# path rather than looked up - so the derivation is what gets pinned hardest here, against the +# live list of published slugs in Config/DocsPublishedPages.txt. That check caught a rule that is +# not guessable from a path: a folder with no README.md is a grouping folder and GitBook drops it +# from the URL entirely, which silently moved seven pages up a level. + +BeforeAll { + $BackendRoot = Split-Path -Parent (Split-Path -Parent (Split-Path -Parent $PSCommandPath)) + $script:BackendRoot = $BackendRoot + $script:RepoRoot = Split-Path -Parent $BackendRoot + $script:DocsRoot = Join-Path $script:RepoRoot 'docs' + + Add-Type -Path (Join-Path $BackendRoot 'Shared/CIPPSharp/bin/CIPPSharp.dll') -ErrorAction SilentlyContinue + + $McpRoot = Join-Path $BackendRoot 'Modules/CIPPCore/Public/MCP' + foreach ($Leaf in 'Get-CippDocLink.ps1', 'ConvertTo-CippDocToken.ps1', 'ConvertFrom-CippDocMarkdown.ps1', + 'Get-CippDocsIndex.ps1', 'Find-CippDoc.ps1') { + . (Join-Path $McpRoot $Leaf) + } + + $env:CIPPRootPath = $BackendRoot + + # Folders owning a README contribute a URL segment; the rest are elided. + $script:SectionFolder = [System.Collections.Generic.HashSet[string]]::new([StringComparer]::OrdinalIgnoreCase) + foreach ($Readme in [System.IO.Directory]::EnumerateFiles($script:DocsRoot, 'README.md', [System.IO.SearchOption]::AllDirectories)) { + $Dir = [System.IO.Path]::GetDirectoryName($Readme) + if ($Dir.Length -le $script:DocsRoot.Length) { continue } + $script:SectionFolder.Add(($Dir.Substring($script:DocsRoot.Length).TrimStart('\', '/') -replace '\\', '/')) | Out-Null + } + + function Resolve-TestLink { + param([string]$Path, [string]$Heading) + return Get-CippDocLink -RelativePath $Path -Heading $Heading -SectionFolder $script:SectionFolder + } +} + +Describe 'Get-CippDocAnchor' { + + It 'slugifies a heading the way GitBook does' { + Get-CippDocAnchor -Heading 'Self-Hosted Deployment' | Should -Be 'self-hosted-deployment' + } + + It 'deletes apostrophes rather than turning them into separators' { + # 'you-ve-met' would be a dead anchor on every page with a contraction in a heading. + Get-CippDocAnchor -Heading "Confirm You've Met All Prerequisites" | Should -Be 'confirm-youve-met-all-prerequisites' + Get-CippDocAnchor -Heading "Confirm You$([char]0x2019)ve Met All Prerequisites" | Should -Be 'confirm-youve-met-all-prerequisites' + } + + It 'strips leading hashes and inline markdown' { + Get-CippDocAnchor -Heading '### Open the **Management** Portal' | Should -Be 'open-the-management-portal' + Get-CippDocAnchor -Heading 'See [the guide](https://example.com)' | Should -Be 'see-the-guide' + } + + It 'returns empty for no heading' { + Get-CippDocAnchor -Heading '' | Should -BeNullOrEmpty + } +} + +Describe 'Get-CippDocLink' { + + It 'maps a page to its published URL and GitHub source' { + $Link = Resolve-TestLink -Path 'setup/setting-up-cipp/install.md' + $Link.docsUrl | Should -Be 'https://docs.cipp.app/setup/setting-up-cipp/install' + $Link.githubUrl | Should -Be 'https://github.com/CyberDrain/CIPP/blob/dev/docs/setup/setting-up-cipp/install.md' + } + + It 'collapses a README to the folder it indexes' { + (Resolve-TestLink -Path 'setup/setting-up-cipp/README.md').docsUrl | + Should -Be 'https://docs.cipp.app/setup/setting-up-cipp' + } + + It 'publishes the docs root README at /readme' { + (Resolve-TestLink -Path 'README.md').docsUrl | Should -Be 'https://docs.cipp.app/readme' + } + + It 'elides a folder that owns no README' { + # docs/user-documentation/email/resources has no README, so GitBook drops the segment. + (Resolve-TestLink -Path 'user-documentation/email/resources/management/equipment/edit.md').docsUrl | + Should -Be 'https://docs.cipp.app/user-documentation/email/management/equipment/edit' + } + + It 'keeps a top-level folder even though it owns no README' { + # 'setup' is a SUMMARY.md '## Group', which always contributes its slug. + $script:SectionFolder.Contains('setup') | Should -BeFalse + (Resolve-TestLink -Path 'setup/installation/owntenant.md').docsUrl | + Should -Be 'https://docs.cipp.app/setup/installation/owntenant' + } + + It 'keeps the real repo path in the GitHub link even when the docs URL elides a folder' { + $Link = Resolve-TestLink -Path 'user-documentation/email/resources/management/equipment/edit.md' + $Link.githubUrl | Should -Match 'docs/user-documentation/email/resources/management/equipment/edit\.md$' + } + + It 'derives the CIPP route for a user-documentation page' { + (Resolve-TestLink -Path 'user-documentation/identity/administration/users/README.md').appPath | + Should -Be '/identity/administration/users' + } + + It 'gives no route to pages that do not document a screen' { + (Resolve-TestLink -Path 'setup/setting-up-cipp/install.md').appPath | Should -BeNullOrEmpty + } + + It 'appends the heading anchor to both links' { + $Link = Resolve-TestLink -Path 'setup/setting-up-cipp/install.md' -Heading 'Self-Hosted Deployment' + $Link.docsUrl | Should -Be 'https://docs.cipp.app/setup/setting-up-cipp/install#self-hosted-deployment' + $Link.githubUrl | Should -Match '#self-hosted-deployment$' + } + + It 'derives a URL matching the live site for every published page' { + # The whole-corpus check. Config/DocsPublishedPages.txt is a snapshot of docs.cipp.app's + # own llms.txt index, so a mismatch here means the tool would hand out a URL that 404s. + $Snapshot = Join-Path $env:CIPPRootPath 'Config/DocsPublishedPages.txt' + $Snapshot | Should -Exist + + $Published = [System.Collections.Generic.HashSet[string]]::new([StringComparer]::OrdinalIgnoreCase) + foreach ($Line in (Get-Content $Snapshot)) { + $Trimmed = $Line.Trim() + if ($Trimmed -and -not $Trimmed.StartsWith('#')) { $Published.Add($Trimmed) | Out-Null } + } + + $Generated = [System.Collections.Generic.HashSet[string]]::new([StringComparer]::OrdinalIgnoreCase) + foreach ($File in [System.IO.Directory]::EnumerateFiles($script:DocsRoot, '*.md', [System.IO.SearchOption]::AllDirectories)) { + $Rel = $File.Substring($script:DocsRoot.Length).TrimStart('\', '/') -replace '\\', '/' + if ($Rel -eq 'SUMMARY.md' -or $Rel -like '.gitbook/*' -or $Rel -like 'legacy-setup-hidden-from-nav/*') { continue } + $Generated.Add((Resolve-TestLink -Path $Rel).slug) | Out-Null + } + + $Unreachable = @($Published | Where-Object { -not $Generated.Contains($_) }) + $Unreachable | Should -BeNullOrEmpty -Because "every published page must be reachable from a repo path; missing: $($Unreachable -join ', ')" + } +} + +Describe 'ConvertFrom-CippDocMarkdown' { + + It 'takes the title from the H1 and the description from frontmatter' { + $Parsed = ConvertFrom-CippDocMarkdown -RelativePath 'a/b.md' -Markdown @' +--- +description: Installing Your CIPP +--- + +# Installation + +Intro prose. + +## First Section + +Body text. +'@ + $Parsed.Title | Should -Be 'Installation' + $Parsed.Description | Should -Be 'Installing Your CIPP' + $Parsed.Chunks.Count | Should -Be 2 + $Parsed.Chunks[0].Heading | Should -BeNullOrEmpty + $Parsed.Chunks[1].Heading | Should -Be 'First Section' + } + + It 'strips GitBook block tags and HTML embeds' { + $Parsed = ConvertFrom-CippDocMarkdown -RelativePath 'a/b.md' -Markdown @' +# Page + +{% stepper %} +{% step %} +Real content here. +{% endstep %} +{% endstepper %} + +
+'@ + $Parsed.Chunks[0].Text | Should -Match 'Real content here' + $Parsed.Chunks[0].Text | Should -Not -Match 'stepper' + $Parsed.Chunks[0].Text | Should -Not -Match 'figure' + } + + It 'does not split a page on a comment inside a fenced code block' { + # A '# Install the module' comment in a PowerShell sample is not a heading. + $Parsed = ConvertFrom-CippDocMarkdown -RelativePath 'a/b.md' -Markdown @' +# Page + +## Real Section + +```powershell +# Install the module +Install-Module Foo +``` +'@ + @($Parsed.Chunks | Where-Object { $_.Heading }).Count | Should -Be 1 + $Parsed.Chunks[-1].Text | Should -Match 'Install-Module Foo' + } + + It 'keeps a link label but discards its URL' { + $Parsed = ConvertFrom-CippDocMarkdown -RelativePath 'a/b.md' -Markdown @' +# Page + +See the [Offboarding Wizard](https://example.com/some-unrelated-slug). +'@ + $Parsed.Chunks[0].Text | Should -Match 'Offboarding Wizard' + $Parsed.Chunks[0].Text | Should -Not -Match 'unrelated-slug' + } + + It 'falls back to the folder name when a README has no H1' { + (ConvertFrom-CippDocMarkdown -RelativePath 'user-documentation/gdap-management/README.md' -Markdown 'Just prose.').Title | + Should -Be 'Gdap Management' + } +} + +Describe 'ConvertTo-CippDocToken' { + + It 'lowercases, drops stop words and stems plurals' { + ConvertTo-CippDocToken -Text 'The Standards are here' | Should -Be @('standard') + } + + It 'indexes a compound identifier whole and in parts' { + $Tokens = ConvertTo-CippDocToken -Text 'ListUsers' + $Tokens | Should -Contain 'listuser' + $Tokens | Should -Contain 'list' + $Tokens | Should -Contain 'user' + } + + It 'splits a dotted role name without losing the full string' { + # 'ReadWrite' splits again on the camel-case boundary, so the parts are read + write. + $Tokens = ConvertTo-CippDocToken -Text 'Identity.User.ReadWrite' + $Tokens | Should -Contain 'identity.user.readwrite' + $Tokens | Should -Contain 'identity' + $Tokens | Should -Contain 'read' + $Tokens | Should -Contain 'write' + } + + It 'tokenises a query and the indexed text identically' { + # If these ever diverge, a term indexed one way is unfindable the other. + (ConvertTo-CippDocToken -Text 'Conditional Access Policies') | + Should -Be (ConvertTo-CippDocToken -Text 'conditional access policy') + } + + It 'returns nothing for empty input' { + ConvertTo-CippDocToken -Text '' | Should -BeNullOrEmpty + } +} + +Describe 'Get-CippDocsRoot' { + + AfterEach { + $env:CIPPDocsPath = $null + $env:CIPPRootPath = $script:BackendRoot + } + + It 'finds the docs in a source checkout' { + (Resolve-Path (Get-CippDocsRoot)).Path.TrimEnd('\', '/') | + Should -Be (Resolve-Path $script:DocsRoot).Path.TrimEnd('\', '/') + } + + It 'prefers CIPPDocsPath when it is set' { + $env:CIPPDocsPath = $script:DocsRoot + Get-CippDocsRoot | Should -Be $script:DocsRoot + } + + It 'skips a directory that exists but holds no markdown' { + # Docker leaves exactly this behind: bind-mounting the docs at /app/API/Docs creates an + # empty backend/Docs on the host, which a bare existence check accepts and then indexes + # to zero pages - every search silently returns nothing. + $Empty = Join-Path ([System.IO.Path]::GetTempPath()) ([guid]::NewGuid().ToString()) + New-Item -ItemType Directory -Path $Empty | Out-Null + try { + $env:CIPPDocsPath = $Empty + Get-CippDocsRoot | Should -Not -Be $Empty + (Resolve-Path (Get-CippDocsRoot)).Path.TrimEnd('\', '/') | + Should -Be (Resolve-Path $script:DocsRoot).Path.TrimEnd('\', '/') + } finally { + Remove-Item -LiteralPath $Empty -Recurse -Force + } + } + + It 'returns null when nothing holds documentation' { + $env:CIPPDocsPath = $null + $env:CIPPRootPath = [System.IO.Path]::GetTempPath() + Get-CippDocsRoot | Should -BeNullOrEmpty + } +} + +Describe 'Find-CippDoc' { + + BeforeAll { + [CIPP.DocsIndex]::Clear() + $null = Get-CippDocsIndex -DocsRoot $script:DocsRoot -Force + } + + It 'builds an index over the shipped docs' { + [CIPP.DocsIndex]::PageCount | Should -BeGreaterThan 300 + [CIPP.DocsIndex]::ChunkCount | Should -BeGreaterThan 1000 + } + + It 'excludes the superseded legacy tree and GitBook includes' { + $Paths = @([CIPP.DocsIndex]::GetPages() | ForEach-Object { $_.RelativePath }) + @($Paths | Where-Object { $_ -like 'legacy-setup-hidden-from-nav/*' }) | Should -BeNullOrEmpty + @($Paths | Where-Object { $_ -like '.gitbook/*' }) | Should -BeNullOrEmpty + } + + It 'withholds a docs URL from pages GitBook does not publish' { + # These exist in docs/ and in SUMMARY.md but are not live, so linking to them would 404. + $Unpublished = @([CIPP.DocsIndex]::GetPages() | Where-Object { -not $_.Published }) + $Unpublished.Count | Should -BeGreaterThan 0 + foreach ($Page in $Unpublished) { + $Page.DocsUrl | Should -BeNullOrEmpty + $Page.GitHubUrl | Should -Not -BeNullOrEmpty + } + } + + It 'finds a page by its own vocabulary' { + $Result = Find-CippDoc -Query 'offboarding wizard' -Limit 5 + @($Result.results | ForEach-Object { $_.path }) | Should -Contain 'user-documentation/identity/administration/offboarding-wizard.md' + } + + It 'reaches conditional access from the abbreviation via synonym expansion' { + # 'CA' appears nowhere in the prose of the pages this has to find. + $Result = Find-CippDoc -Query 'CA policy' -Limit 5 + @($Result.results | ForEach-Object { $_.title }) -join ' ' | Should -Match 'Conditional Access|CA Polic|CA Template' + } + + It 'recovers from typos' { + $Result = Find-CippDoc -Query 'conditonal acces' -Limit 5 + $Result.matchCount | Should -BeGreaterThan 0 + @($Result.results | ForEach-Object { $_.title }) -join ' ' | Should -Match 'Conditional|CA ' + } + + It 'deep-links to the matching section' { + $Result = Find-CippDoc -Query 'offboarding options' -Limit 5 + $Hit = @($Result.results | Where-Object { $_.section -and $_.docsUrl -match '#' })[0] + $Hit | Should -Not -BeNullOrEmpty + $Hit.docsUrl | Should -Match '#' + } + + It 'looks up documentation by CIPP route' { + $Result = Find-CippDoc -Path '/identity/administration/users' -Limit 5 + $Result.matchCount | Should -BeGreaterThan 0 + @($Result.results | ForEach-Object { $_.path }) | Should -Contain 'user-documentation/identity/administration/users/README.md' + } + + It 'scopes a keyword search to a route' { + $Result = Find-CippDoc -Query 'permissions' -Path '/identity/administration/users' -Limit 5 + $Result.matchCount | Should -BeGreaterThan 0 + foreach ($Hit in $Result.results) { + $Hit.path | Should -BeLike 'user-documentation/identity/administration/users*' + } + } + + It 'does not fill the results with one page' { + # Without a per-page cap a single long page crowds out every other answer. + # Grouped through a script block deliberately: results are ordered hashtables, and + # Group-Object -Property path does not resolve a hashtable key - it silently returns + # one group of everything, which reads as a failing cap when the cap is fine. + $Result = Find-CippDoc -Query 'user' -Limit 8 + $Counts = @($Result.results | Group-Object -Property { $_.path } | ForEach-Object { $_.Count }) + ($Counts | Measure-Object -Maximum).Maximum | Should -BeLessOrEqual 2 + } + + It 'returns nothing rather than noise for a nonsense query' { + (Find-CippDoc -Query 'zzzqqqxyz wibblefrotz' -Limit 5).matchCount | Should -Be 0 + } + + It 'asks for input when given neither query nor path' { + (Find-CippDoc -Limit 5).error | Should -Not -BeNullOrEmpty + } + + It 'reports an unmatched path instead of silently searching everything' { + $Result = Find-CippDoc -Path '/no/such/route' -Limit 5 + $Result.matchCount | Should -Be 0 + $Result.hint | Should -Match 'No documentation page matches' + } + + It 'never returns a result without a usable link' { + foreach ($Hit in (Find-CippDoc -Query 'standards drift' -Limit 8).results) { + ($Hit.docsUrl ?? $Hit.githubUrl) | Should -Not -BeNullOrEmpty + } + } +} + +Describe 'Get-CippDoc' { + + BeforeAll { + [CIPP.DocsIndex]::Clear() + $null = Get-CippDocsIndex -DocsRoot $script:DocsRoot -Force + } + + It 'accepts a repo path, a published slug or a CIPP route' { + foreach ($Identifier in 'user-documentation/identity/administration/users/README.md', + 'user-documentation/identity/administration/users', + '/identity/administration/users') { + $Doc = Get-CippDoc -Path $Identifier + $Doc.error | Should -BeNullOrEmpty -Because "'$Identifier' should resolve" + $Doc.title | Should -Be 'Users' + $Doc.content | Should -Not -BeNullOrEmpty + } + } + + It 'returns the page text with its headings' { + $Doc = Get-CippDoc -Path 'setup/setting-up-cipp/install.md' + $Doc.content | Should -Match 'Self-Hosted Deployment' + $Doc.sections | Should -Contain 'Self-Hosted Deployment' + } + + It 'suggests alternatives for an unknown page, without repeating one' { + $Doc = Get-CippDoc -Path 'no/such/page' + $Doc.error | Should -Not -BeNullOrEmpty + $Doc.suggestions.Count | Should -Be @($Doc.suggestions | Select-Object -Unique).Count + } +} + +Describe 'MCP gateway exposes the docs tools' { + + BeforeAll { + $McpRoot = Join-Path (Split-Path -Parent (Split-Path -Parent (Split-Path -Parent $PSCommandPath))) 'Modules/CIPPCore/Public/MCP' + . (Join-Path $McpRoot 'Get-CippMcpToolList.ps1') + + function Get-CippMcpToolCatalog { return @() } + } + + It 'advertises SearchDocs and GetDoc alongside the API gateway tools' { + $Names = @((Get-CippMcpToolList -Request ([pscustomobject]@{ Query = @{} }) -InformationAction SilentlyContinue) | ForEach-Object { $_.name }) + $Names | Should -Contain 'SearchDocs' + $Names | Should -Contain 'GetDoc' + $Names | Should -Contain 'SearchTools' + } + + It 'gives GetDoc a required path parameter' { + $Tool = @((Get-CippMcpToolList -Request ([pscustomobject]@{ Query = @{} }) -InformationAction SilentlyContinue) | Where-Object { $_.name -eq 'GetDoc' })[0] + $Tool.inputSchema.required | Should -Contain 'path' + } +} diff --git a/build/Dockerfile b/build/Dockerfile index c46172f0a1..a48fad503c 100644 --- a/build/Dockerfile +++ b/build/Dockerfile @@ -79,6 +79,24 @@ COPY build/tools/build-function-parameters.ps1 /build/build-function-parameters. COPY backend/Modules/CIPPCore /gen/CIPPCore RUN pwsh -File /build/build-function-parameters.ps1 -ModulePath /gen/CIPPCore -OutputPath /gen/function-parameters.json +# ── Published docs page list ── +# Config/DocsPublishedPages.txt tells the SearchDocs MCP tool which pages are actually live on +# docs.cipp.app, so a result never carries a URL that 404s. A page can sit in docs/ unpublished, +# and only llms.txt knows which, so this is refreshed from the live site at build time — an image +# then reflects what was published when it was built rather than whenever someone last ran the +# script by hand. The committed copy is the fallback and is what ships if the fetch fails. +# +# ARG BUILD_DATE is here purely to bust the layer cache: the inputs are otherwise just the script +# and the committed list, so without it Docker would reuse a cached layer and ship a stale list +# from an earlier build despite this stage existing. It sits in its own tiny stage, so unlike the +# version args near the frontend build it invalidates nothing expensive. +FROM ps-base AS build-docspages +COPY build/tools/Update-DocsPublishedPages.ps1 /build/Update-DocsPublishedPages.ps1 +COPY backend/Config/DocsPublishedPages.txt /gen/DocsPublishedPages.txt +ARG BUILD_DATE=unknown +RUN echo "Refreshing published docs list (build ${BUILD_DATE})" && \ + pwsh -File /build/Update-DocsPublishedPages.ps1 -OutputPath /gen/DocsPublishedPages.txt -AllowFallback + # ── OpenAPI spec ── # Config/openapi.json, derived from the CIPPHTTP entrypoint sources via AST parse. # Not documentation-only: Get-CippMcpSpec loads it at runtime and Get-CippMcpToolList @@ -184,6 +202,10 @@ COPY --from=build-cippsharp /out/CIPPSharp.dll /app/API/Shared/CIPPSharp/bin/CIP COPY backend/AddChocoApp /app/API/AddChocoApp/ COPY backend/AddMSPApp /app/API/AddMSPApp/ COPY backend/ExecMaintenanceScripts /app/API/ExecMaintenanceScripts/ +# GitBook source for docs.cipp.app, indexed at runtime by the SearchDocs MCP tool +# (Get-CippDocsIndex reads $env:CIPPRootPath/Docs). Markdown only — Dockerfile.dockerignore +# keeps the .gitbook image tree out. ~2 MB, and it version-matches the docs to this build. +COPY docs /app/API/Docs/ # Compiled modules — ordered by change frequency (least → most) COPY --from=build-cippextensions /src/CippExtensions/ /app/API/Modules/CippExtensions/ @@ -199,6 +221,9 @@ COPY --chown=app:app --from=build-functionparams /gen/function-parameters.json / # generated from CIPPHTTP; overwrites the committed copy so the image always ships # a spec that matches the entrypoints beside it COPY --chown=app:app --from=build-openapi /gen/openapi.json /app/API/Config/openapi.json +# refreshed from docs.cipp.app at build time; overwrites the committed fallback so the +# SearchDocs tool links against what was actually published when this image was built +COPY --chown=app:app --from=build-docspages /gen/DocsPublishedPages.txt /app/API/Config/DocsPublishedPages.txt # Frontend — changes most frequently, placed last COPY --from=frontend-build /build/out/ /app/Frontend/ diff --git a/build/Dockerfile.dockerignore b/build/Dockerfile.dockerignore index 40b24aeafc..3d61049381 100644 --- a/build/Dockerfile.dockerignore +++ b/build/Dockerfile.dockerignore @@ -22,11 +22,22 @@ frontend/node_modules/ frontend/.next/ frontend/out/ -# Non-runtime +# Non-runtime. NOTE: `*.md` matches only markdown in the context ROOT - a dockerignore `*` does +# not cross '/'. Nested markdown that is read at runtime (backend/Modules/CIPPTests, which +# Invoke-ListTests.ps1 enumerates, and docs/, which the SearchDocs MCP tool indexes) is therefore +# untouched by this rule. Do not "fix" it to `**/*.md`. +# +# Also note this file, not the context-root .dockerignore, is what BuildKit applies: CI builds +# with `context: .` and `file: build/Dockerfile`, and a `.dockerignore` overrides the +# one at the context root. *.md .env build/.env.example +# Docs are shipped for the SearchDocs MCP tool, but only the markdown - the .gitbook tree is +# ~437 KB of images that are never indexed. +docs/.gitbook/ + # Knowledge graphs (local dev tooling) graphify/ graphify-out/ diff --git a/build/Dockerfile.release b/build/Dockerfile.release index 447331ba95..e6f8cf5210 100644 --- a/build/Dockerfile.release +++ b/build/Dockerfile.release @@ -79,6 +79,24 @@ COPY build/tools/build-function-parameters.ps1 /build/build-function-parameters. COPY backend/Modules/CIPPCore /gen/CIPPCore RUN pwsh -File /build/build-function-parameters.ps1 -ModulePath /gen/CIPPCore -OutputPath /gen/function-parameters.json +# ── Published docs page list ── +# Config/DocsPublishedPages.txt tells the SearchDocs MCP tool which pages are actually live on +# docs.cipp.app, so a result never carries a URL that 404s. A page can sit in docs/ unpublished, +# and only llms.txt knows which, so this is refreshed from the live site at build time — an image +# then reflects what was published when it was built rather than whenever someone last ran the +# script by hand. The committed copy is the fallback and is what ships if the fetch fails. +# +# ARG BUILD_DATE is here purely to bust the layer cache: the inputs are otherwise just the script +# and the committed list, so without it Docker would reuse a cached layer and ship a stale list +# from an earlier build despite this stage existing. It sits in its own tiny stage, so unlike the +# version args near the frontend build it invalidates nothing expensive. +FROM ps-base AS build-docspages +COPY build/tools/Update-DocsPublishedPages.ps1 /build/Update-DocsPublishedPages.ps1 +COPY backend/Config/DocsPublishedPages.txt /gen/DocsPublishedPages.txt +ARG BUILD_DATE=unknown +RUN echo "Refreshing published docs list (build ${BUILD_DATE})" && \ + pwsh -File /build/Update-DocsPublishedPages.ps1 -OutputPath /gen/DocsPublishedPages.txt -AllowFallback + # ── OpenAPI spec ── # Config/openapi.json, derived from the CIPPHTTP entrypoint sources via AST parse. # Not documentation-only: Get-CippMcpSpec loads it at runtime and Get-CippMcpToolList @@ -184,6 +202,10 @@ COPY --from=build-cippsharp /out/CIPPSharp.dll /app/API/Shared/CIPPSharp/bin/CIP COPY backend/AddChocoApp /app/API/AddChocoApp/ COPY backend/AddMSPApp /app/API/AddMSPApp/ COPY backend/ExecMaintenanceScripts /app/API/ExecMaintenanceScripts/ +# GitBook source for docs.cipp.app, indexed at runtime by the SearchDocs MCP tool +# (Get-CippDocsIndex reads $env:CIPPRootPath/Docs). Markdown only — the .dockerignore keeps the +# .gitbook image tree out. ~2 MB, and it version-matches the docs to this build. +COPY docs /app/API/Docs/ # Compiled modules — ordered by change frequency (least → most) COPY --from=build-cippextensions /src/CippExtensions/ /app/API/Modules/CippExtensions/ @@ -199,6 +221,9 @@ COPY --chown=app:app --from=build-functionparams /gen/function-parameters.json / # generated from CIPPHTTP; overwrites the committed copy so the image always ships # a spec that matches the entrypoints beside it COPY --chown=app:app --from=build-openapi /gen/openapi.json /app/API/Config/openapi.json +# refreshed from docs.cipp.app at build time; overwrites the committed fallback so the +# SearchDocs tool links against what was actually published when this image was built +COPY --chown=app:app --from=build-docspages /gen/DocsPublishedPages.txt /app/API/Config/DocsPublishedPages.txt # Frontend — changes most frequently, placed last COPY --from=frontend-build /build/out/ /app/Frontend/ diff --git a/build/docker-compose-all.yml b/build/docker-compose-all.yml index 584fbf4a16..8424e9b6f1 100644 --- a/build/docker-compose-all.yml +++ b/build/docker-compose-all.yml @@ -83,6 +83,9 @@ services: - ASPNETCORE_ENVIRONMENT=Development - CRAFT_VERBOSE=true - App__Setup__Enabled=false + # Where the SearchDocs MCP tool finds the docs. Only needed in dev: the image puts them + # at $CIPPRootPath/Docs, which Get-CippDocsRoot probes by default. + - CIPPDocsPath=/app/Docs - AzureWebJobsStorage=DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite:10000/devstoreaccount1;QueueEndpoint=http://azurite:10001/devstoreaccount1;TableEndpoint=http://azurite:10002/devstoreaccount1; - NonLocalHostAzurite=true - CIPPNG=true @@ -118,6 +121,16 @@ services: # Live backend source. rw because runtime permission extraction writes # Config/function-permissions.json back into the repo (gitignored). - ../backend:/app/API + # GitBook docs, indexed by the SearchDocs MCP tool. Without this the tool reports the + # docs as missing, because only backend/ is bind-mounted and ../docs is not reachable + # from inside the container. + # + # Mounted at /app/Docs and pointed at by CIPPDocsPath rather than at the image's own + # /app/API/Docs: that path sits INSIDE the ../backend bind mount, and Docker creates a + # nested mountpoint on the host to hang it off — leaving an empty backend/Docs in the + # working tree that then shadows the real docs for Pester and anything else running + # outside the container. + - ../docs:/app/Docs:ro # Writable manifest shims, seeded by cipp-manifests. Each mount shadows a single # .psd1 so DevExpandModuleExports rewrites the copy under .devmanifests/ instead of # the tracked file in backend/Modules. Everything else in the module dir (Public/, diff --git a/build/docker-compose-no-frontend.yml b/build/docker-compose-no-frontend.yml index 819293ac1e..167c438eef 100644 --- a/build/docker-compose-no-frontend.yml +++ b/build/docker-compose-no-frontend.yml @@ -48,6 +48,9 @@ services: - ASPNETCORE_ENVIRONMENT=Development - CRAFT_VERBOSE=true - App__Setup__Enabled=false + # Where the SearchDocs MCP tool finds the docs. Only needed in dev: the image puts them + # at $CIPPRootPath/Docs, which Get-CippDocsRoot probes by default. + - CIPPDocsPath=/app/Docs - AzureWebJobsStorage=DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://cipp-azurite:10000/devstoreaccount1;QueueEndpoint=http://cipp-azurite:10001/devstoreaccount1;TableEndpoint=http://cipp-azurite:10002/devstoreaccount1; - NonLocalHostAzurite=true - CIPPNG=true @@ -69,6 +72,16 @@ services: # Live backend source. rw because runtime permission extraction writes # Config/function-permissions.json back into the repo (gitignored). - ../backend:/app/API + # GitBook docs, indexed by the SearchDocs MCP tool. Without this the tool reports the + # docs as missing, because only backend/ is bind-mounted and ../docs is not reachable + # from inside the container. + # + # Mounted at /app/Docs and pointed at by CIPPDocsPath rather than at the image's own + # /app/API/Docs: that path sits INSIDE the ../backend bind mount, and Docker creates a + # nested mountpoint on the host to hang it off — leaving an empty backend/Docs in the + # working tree that then shadows the real docs for Pester and anything else running + # outside the container. + - ../docs:/app/Docs:ro # ModuleBuilder-compiled CIPP modules overlay the bind-mounted source: each # mount shadows just that module's subdir, so the container imports the fast # single-file .psm1 instead of dot-sourcing Public/*.ps1 in every runspace. diff --git a/build/tools/Update-DocsPublishedPages.ps1 b/build/tools/Update-DocsPublishedPages.ps1 new file mode 100644 index 0000000000..f8ed5d5f90 --- /dev/null +++ b/build/tools/Update-DocsPublishedPages.ps1 @@ -0,0 +1,97 @@ +<# +.SYNOPSIS + Refreshes backend/Config/DocsPublishedPages.txt from the live docs.cipp.app index. +.DESCRIPTION + The docs search MCP tool (SearchDocs / GetDoc) links every result back to docs.cipp.app. A + page can exist in docs/ without being published - drafts stay in SUMMARY.md, and the + legacy tree is hidden from nav - so linking straight from the file path would hand callers + URLs that 404. docs.cipp.app/llms.txt is the only authoritative list of what is live, and + this script snapshots its slugs so the check stays offline at runtime. + + Run it when pages are published or unpublished. The diff should be small; a large one usually + means the docs were restructured, in which case check Get-CippDocLink's slug derivation still + holds (Tests/Mcp/CippDocs.Tests.ps1 pins it). + + The container build runs this too (see the build-docspages stage in build/Dockerfile), so an + image always ships the list as it was at build time rather than whenever someone last ran this + by hand. There -AllowFallback is passed: docs.cipp.app being unreachable, or answering with + something that does not parse, must not fail the build. The committed copy is the fallback and + is simply left in place. +.EXAMPLE + pwsh build/tools/Update-DocsPublishedPages.ps1 +.EXAMPLE + pwsh build/tools/Update-DocsPublishedPages.ps1 -OutputPath /gen/DocsPublishedPages.txt -AllowFallback +#> +[CmdletBinding()] +param( + [string]$Uri = 'https://docs.cipp.app/llms.txt', + [string]$OutputPath = (Join-Path $PSScriptRoot '../../backend/Config/DocsPublishedPages.txt'), + + # Warn and keep the existing file instead of throwing. For the container build, where a + # transient network failure must not take the image down with it. + [switch]$AllowFallback +) + +$ErrorActionPreference = 'Stop' + +# $Allowed is passed rather than closed over: the analyzer cannot see a switch used only inside +# a nested function and reports it unused, and being explicit reads better at the call sites. +function Resolve-FetchFailure { + param( + [Parameter(Mandatory)][string]$Message, + [bool]$Allowed + ) + if (-not $Allowed) { throw $Message } + Write-Warning $Message + Write-Host 'Keeping the committed DocsPublishedPages.txt as the fallback.' + # A page missing from a stale list loses its docs.cipp.app link and falls back to GitHub, + # which is a far better outcome than failing the build. + exit 0 +} + +Write-Host "Fetching $Uri ..." +try { + $Response = Invoke-WebRequest -Uri $Uri -UseBasicParsing -TimeoutSec 60 +} catch { + Resolve-FetchFailure -Message "Could not fetch $Uri : $($_.Exception.Message)" -Allowed $AllowFallback +} + +$Slugs = [System.Collections.Generic.SortedSet[string]]::new([StringComparer]::OrdinalIgnoreCase) +foreach ($Line in ($Response.Content -split '\r?\n')) { + if ($Line -match 'https://docs\.cipp\.app/([^\s)]+)') { + $Slug = $Matches[1] -replace '\.md$', '' + # llms.txt also advertises the site's AI-ask URL template, which is not a page. + if ($Slug -match '[?#]') { continue } + $Slugs.Add($Slug) | Out-Null + } +} + +if ($Slugs.Count -lt 100) { + # A captive portal, an error page or a redirect all return 200 with a body that parses to + # nothing useful. Overwriting a good list with that would silently strip every docs link. + Resolve-FetchFailure -Allowed $AllowFallback -Message "Only $($Slugs.Count) slugs parsed from $Uri - refusing to overwrite the snapshot with what looks like a failed fetch." +} + +$Resolved = [System.IO.Path]::GetFullPath($OutputPath) +$Previous = if (Test-Path -LiteralPath $Resolved) { + @(Get-Content -LiteralPath $Resolved | Where-Object { $_ -and -not $_.StartsWith('#') }) +} else { @() } + +$Header = @( + '# Slugs published on docs.cipp.app, snapshotted from llms.txt.' + '# Generated by build/tools/Update-DocsPublishedPages.ps1 - do not hand-edit.' + '# Read by Get-CippDocsPublishedSet so the docs search index never emits a URL that 404s.' + "# $($Slugs.Count) pages." +) +[System.IO.File]::WriteAllLines($Resolved, @($Header) + @($Slugs), (New-Object System.Text.UTF8Encoding $false)) + +$Added = @($Slugs | Where-Object { $_ -notin $Previous }) +$Removed = @($Previous | Where-Object { -not $Slugs.Contains($_) }) + +Write-Host "Wrote $($Slugs.Count) slugs to $Resolved" +if ($Previous.Count -gt 0) { + Write-Host " added: $($Added.Count)" + $Added | Select-Object -First 10 | ForEach-Object { Write-Host " + $_" } + Write-Host " removed: $($Removed.Count)" + $Removed | Select-Object -First 10 | ForEach-Object { Write-Host " - $_" } +}