diff --git a/doc/README.md b/doc/README.md index fabaf146..06514fba 100644 --- a/doc/README.md +++ b/doc/README.md @@ -18,7 +18,7 @@ | [Hive & Data Stores](cli/hive-data.md) | hive, secret, lookup, playbook, note, sop, adapter, cloud-sensor, extension | | [Infrastructure](cli/infrastructure.md) | sync, output, artifact, payload, yara, integrity, logging, exfil | | [Cloud Security & Code Security](cli/cloud-security.md) | cloudsec (findings, inventory, graph, compliance, CAASM, code lane, container images, fleet, exports) | -| [Email Security](cli/email-security.md) | mailsec (triage queue, EML, remediation, campaigns, reports, hunts, rules, tenant purge) | +| [Email Security](cli/email-security.md) | mailsec (onboarding, coverage, triage, EML, remediation, campaigns, reports, rules, tenant purge) | | [Other Commands](cli/other-commands.md) | api, arl, usp, spotcheck, job, schema, completion, help/discover | ## SDK Reference @@ -34,6 +34,7 @@ | [Search & Insight](sdk/search-insight.md) | Search (LCQL), Insight (IOC) | | [Streaming](sdk/streaming.md) | Spout, Firehose | | [Configuration Sync](sdk/configs.md) | Configs (IaC) | +| [Security Products](sdk/security-products.md) | CloudSec, Mailsec, onboarding, and pagination | | [Other Classes](sdk/other-classes.md) | Extensions, Artifacts, Payloads, Outputs, AI, Billing, CloudSec, etc. | ## External Resources diff --git a/doc/cli/README.md b/doc/cli/README.md index 0460e2cc..9a82d758 100644 --- a/doc/cli/README.md +++ b/doc/cli/README.md @@ -115,19 +115,22 @@ limacharlie sensor list -W # All columns untruncated ```bash # List all commands grouped by use-case -limacharlie discover -limacharlie discover --profile detection_engineering -limacharlie discover --profile incident_response +limacharlie help discover +limacharlie help discover --profile cloud_security +limacharlie help discover --profile email_security # Concept guides limacharlie help d&r-rules limacharlie help hive limacharlie help lcql +limacharlie help cloud-security +limacharlie help code-security +limacharlie help email-security # Quick-reference cheat sheets -limacharlie cheatsheet common-operations -limacharlie cheatsheet detection-engineering -limacharlie cheatsheet incident-response +limacharlie help cheatsheet --name cloud-security +limacharlie help cheatsheet --name code-security +limacharlie help cheatsheet --name email-security # Detailed explanation of any command limacharlie dr create --ai-help @@ -147,7 +150,7 @@ limacharlie schema dr create | [Hive & Data Stores](hive-data.md) | hive, secret, lookup, playbook, note, sop, adapter, cloud-sensor, extension | | [Infrastructure](infrastructure.md) | sync, output, artifact, payload, yara, integrity, logging, exfil | | [Cloud Security & Code Security](cloud-security.md) | cloudsec (findings, inventory, graph, compliance, CAASM, code lane, container images, fleet, exports) | -| [Email Security](email-security.md) | mailsec (triage queue, EML, remediation, campaigns, reports, hunts, rules, tenant purge) | +| [Email Security](email-security.md) | mailsec (onboarding, coverage, triage, EML, remediation, campaigns, reports, rules, tenant purge) | | [Other Commands](other-commands.md) | api, arl, usp, spotcheck, job, schema, completion, help/discover, case | ## See Also diff --git a/doc/cli/cloud-security.md b/doc/cli/cloud-security.md index 05eccd49..af70ae6a 100644 --- a/doc/cli/cloud-security.md +++ b/doc/cli/cloud-security.md @@ -2,9 +2,12 @@ # Cloud Security (CNAPP) & Code Security +Install or upgrade with `python -m pip install --upgrade limacharlie`. See [installation](../getting-started.md#installation) for setup. + Commands for the LimaCharlie Cloud Security surface: the merged, risk-ranked findings worklist (CSPM misconfigurations + attack paths + CIEM + code and container-image vulnerabilities), the cloud resource inventory and security graph, compliance assessment (live and audit-grade), the risk overview, CAASM (third-party asset attack surface), the AppSec code lane and container-image inventory, sensor↔cloud-asset resolution, finding triage, CSV exports, and the multi-org fleet overview. -Reads usually require `cloudsec.get`. Most writes require `cloudsec.set`; AutoFix and remediation decisions require `cloudsec.respond`, and reading an IaC map receipt requires `cloudsec.set`. Every command requires the org to be subscribed to the Cloud Security extension: +Reads usually require `cloudsec.get`. Local `code scan` without ingestion and +`code iac-map extract` work offline; they do not require a subscription or API key. Most writes require `cloudsec.set`; AutoFix and remediation decisions require `cloudsec.respond`, and reading an IaC map receipt requires `cloudsec.set`. API commands require the org to be subscribed to the Cloud Security extension: ```bash limacharlie extension subscribe --name ext-cloud-security @@ -14,6 +17,10 @@ Provider credentials and the cloudsec policies are hive records — manage them Every command supports `--ai-help` for a detailed description with examples. +For `code pr-check`, GitHub requires `--base-sha`; GitLab.com and Bitbucket +Cloud may omit it because the provider resolves the base. Supply `--head-sha` +and `--action` for every provider. `--action edited` is GitHub-only. + ## Overview & posture ```bash @@ -249,7 +256,21 @@ Tenant → management group → subscription → resource group → resource con ## CSV exports -The server walks the full filtered set (no pagination), capped at 100k rows; a trailing `#` comment row marks a truncated export. +By default the server walks the full filtered set, capped at 100k rows; a trailing +`#` comment row marks a truncated export. For findings and inventory, use +`--max-rows` to export in bounded requests. The size is rounded up to full +1000-row pages. If more rows remain, the CSV ends with `# next_cursor=`; +pass that token to the next request with the same size, filters and sort. A chunk +without a continuation comment is the end. Each chunk includes its own header. + +```bash +limacharlie cloudsec export findings --max-rows 2000 -o findings-1.csv +limacharlie cloudsec export findings --max-rows 2000 --cursor "" -o findings-2.csv +``` + +`--cursor` requires `--max-rows` so a resumed request cannot silently restart the +export. CSV comments can also report truncation or a mid-stream error; inspect +them before treating an export as complete. ```bash limacharlie cloudsec export findings -o findings.csv --severity CRITICAL @@ -270,12 +291,22 @@ limacharlie cloudsec resolve assets "lcrn:...instance/web-1" # asset -> sensors ```bash limacharlie cloudsec caasm assets -q laptop --limit 50 +limacharlie cloudsec caasm assets --kind device --source ms_graph --posture-encryption "" +limacharlie cloudsec caasm assets --sort last_seen limacharlie cloudsec caasm coverage --status open --severity HIGH limacharlie cloudsec caasm policy get limacharlie cloudsec caasm policy set --input-file policy.yaml limacharlie cloudsec caasm ingest --source okta --records-file users.json ``` +Asset selectors `--kind`, `--source`, `--posture-encryption`, +`--posture-screen-lock`, `--posture-compromised` and `--posture-managed` are +repeatable: values within a selector are OR'd, selectors are AND'd. Use the posture +values your sources report; an empty value selects assets where no source reported +that fact. Unreported posture never means compliant. `--sort urn` is the stable +walk order; `--sort last_seen` shows the newest observations first. Follow +`next_cursor` until absent, including after short pages. + Ingest sources today: `sentinelone`, `crowdstrike`, `defender`, `okta`, `entraid`, `ms_graph`, `wiz` (the registry grows and is validated server-side). ## Providers @@ -284,8 +315,20 @@ Ingest sources today: `sentinelone`, `crowdstrike`, `defender`, `okta`, `entraid limacharlie cloudsec provider test --input-file provider.yaml # credential preflight (ephemeral) limacharlie cloudsec provider manifest # coverage manifests, all providers limacharlie cloudsec provider manifest --type gcp +limacharlie cloudsec provider m365-certificate my-entra --client-id "" --out connection.cer ``` +For Entra/Microsoft 365 certificate authentication, `m365-certificate` requires +both `cloudsec.set` and `secret.set`. It stores the private key in the organization's +secret store and returns only the public certificate and a `credentials` Hive +reference. Upload `connection.cer` under **Certificates & secrets → Certificates** +in your Entra app registration, then use the returned reference as `credentials` +in the provider record. Grant the app the provider permissions before running +`provider test`. Repeating generation returns the same certificate. +`--replace` replaces the stored key pair immediately. An existing connection may +stop authenticating until you upload the replacement public certificate to the +Entra app registration. + Saved provider configs live in the `cloudsec_provider` hive: ```bash @@ -372,7 +415,11 @@ limacharlie hive set --hive-name cloudsec_policy --key hub-private-image \ `registry` is `dockerhub`, `quay` or `ghcr`; `repository` is one lowercase `namespace/image` (a GHCR path may be deeper), `username` a registry account or robot name, and `secret_ref` must be `hive://secret/` of an existing secret you can read. Use a read-only token. ECR and ACR images use the cloud connection's own read access instead (`setup_path` `integrations/`), scoped to one repository per pull. -`code capabilities` covers **GitHub connections only** — a GitLab or Bitbucket connection scans with its own read-only token and has no write plane to detect, so it never appears, not even as `unknown`. Use `provider manifest` for those. A capability of `available` means the control MAY be offered, not that anything fires on its own. +`code capabilities` reports GitHub connections. GitLab.com and Bitbucket Cloud +connections also appear when their workflow support is enabled in your deployment; +an absent connection does not mean repository scanning is off. Use +`provider manifest` for collection coverage. A capability of `available` means the +control can be offered, not that anything fires on its own. `code fixes` pages differently from the rest of cloudsec: backend default 5, max 20, not the shared 1000-row cap. @@ -429,9 +476,10 @@ not a `/cloudsec` API route. It requires `ext.request`. The SDK equivalent is ```bash limacharlie cloudsec image repos --with-findings --sort risk -limacharlie cloudsec image repo-facets +limacharlie cloudsec image repo-facets --lineage-facet limacharlie cloudsec image list --findings with --running --sort risk limacharlie cloudsec image list --tag latest --registry gcr.io --all +limacharlie cloudsec image list --lineage-status unknown --lineage-status ambiguous --all limacharlie cloudsec image get sha256:<64 hex> limacharlie cloudsec finding list --image-urn "" ``` @@ -440,6 +488,14 @@ An image is keyed on its **digest alone**, so one row is the same artifact every `repositories`, `memberships`, `workloads` and `source_repositories` are BOUNDED SAMPLES of 100 with no pagination — the paired `*_count` is the truth, and only memberships carry a `_truncated` flag. To get past 100 placements, use `image list --repo-urn ...` instead. +`image list --lineage-status` accepts repeatable `verified`, `asserted`, +`inferred`, `ambiguous` and `unknown` selectors. They select the effective +source-lineage state, separately from image-signature status (`--signed`). A stale +decision counts as `unknown`. The SDK checks `applied_lineage_status`; an older +server that cannot acknowledge the filter raises an error instead of returning an +unfiltered page. `image repo-facets --lineage-facet` adds exact digest-global +`lineage_statuses` counts; repository selectors do not narrow those counts. + `image get` also returns a digest-bound `lineage` decision. Read its `tier` (`inferred`, `tool_emitted`, or `our_signed_push`), `status`, and `reason` together: `inferred` and `asserted` do not mean a verified build. The @@ -502,7 +558,11 @@ A rule set document holds one entry per rule file, where each `rules` value is a Before the scan starts, the CLI refuses a document the scanner would not accept: unknown fields, a version other than 1, a record without a unique non-empty `key` or a `rules` object, more than 32 MiB, or no rules at all. The scanner checks each rule, then reports and skips any rule it cannot load. -**Scanner version.** The default image is pinned to scanner v0.16.0. A `--image` or `--binary` running `sast` must be v0.16.0 or newer, because older scanners reject the rule-set flags. That failure is a usage error (exit 2), and the CLI's error message names the version you need. A scan without `sast` passes no rule-set flag, so it still runs on older scanners. +**Scanner version and access.** The default image is pinned to scanner v0.24.0. +Pulling the default image requires registry access. If it is unavailable to your +account, use an accessible scanner image with `--image` or an installed +`scanner-agent` with `--binary`; do not assume an organization API key grants +container-registry access. A `--image` or `--binary` running `sast` must be v0.16.0 or newer, because older scanners reject the rule-set flags. That failure is a usage error (exit 2), and the CLI's error message names the version you need. A scan without `sast` passes no rule-set flag, so it still runs on older scanners. ### Sanitized IaC maps diff --git a/doc/cli/email-security.md b/doc/cli/email-security.md index 00b298e4..66b87cb7 100644 --- a/doc/cli/email-security.md +++ b/doc/cli/email-security.md @@ -2,6 +2,8 @@ # Email Security +Install or upgrade with `python -m pip install --upgrade limacharlie`. See [installation](../getting-started.md#installation) for setup. + Commands for the LimaCharlie Email Security surface: mailbox coverage, the message triage queue and its drawer, the justified raw-EML download, analyst verdict revision, per-message and bulk remediation at the provider, campaigns, sender profiles, the action audit trail, the abuse-mailbox report queue, standalone EML analysis, custom-rule validation and backtest, the connection preflight, and the tenant purge. Four permissions rather than the usual get/set pair, because the product asks to be trusted with four different things: @@ -10,8 +12,15 @@ Four permissions rather than the usual get/set pair, because the product asks to |---|---| | `mailsec.get` | Read the product's own view: the queue, the drawer, campaigns, senders, the audit trail | | `mailsec.set` | Change detection behaviour and triage state | -| `mailsec.act` | Remediate live mail at the provider | -| `mailsec.get.eml` | Take the original bytes of somebody's mail out of the building; requires a logged justification | +| `mailsec.act` | Remediate live mail, revise verdicts and test provider connections | +| `mailsec.get.eml` | Download original message bytes; also requires `mailsec.get` and a logged justification | + +Connection testing and verdict revision require `mailsec.act`. Connection records +have separate `mailsec_provider.*` permissions: creating a provider and enabling +its record require `mailsec_provider.set` and `mailsec_provider.set.mtd`. +Policy and `dr-mail` writes use `mailsec.set`. Creating and enabling the credential +secret requires `secret.set` and `secret.set.mtd`. Report resolution and reopening +use `mailsec.set`. `mailsec tenant purge` is the exception: it is Owner-level, and needs `mailsec.act` **and** `billing.ctrl` **and** `user.ctrl` — the same trio `org delete` asks for. @@ -27,16 +36,74 @@ Every command supports `--ai-help` for a detailed description with examples. ## Coverage & onboarding +Start by selecting your organization with the global `--oid` option, or your +configured default. Subscribe to the extension, then fetch the provider's current +setup guide. For Workspace, supplying your Google Cloud project and service account +fills those values into the returned commands; reading the guide creates no resources. + ```bash limacharlie mailsec coverage --window-days 30 # mailboxes protected vs not +limacharlie mailsec coverage --since 2026-09-01T00:00:00Z --until 2026-09-02T00:00:00Z limacharlie mailsec onboarding --provider gworkspace +limacharlie mailsec onboarding --provider gworkspace --project-id your-project --sa-email mailsec@your-project.iam.gserviceaccount.com limacharlie mailsec onboarding --provider m365 -limacharlie mailsec connection test gws-exp # post-save credential preflight -limacharlie mailsec connection test gws-exp --include-watch +limacharlie mailsec connection test workspace-mail # post-save credential preflight +limacharlie mailsec connection test workspace-mail --include-watch ``` `coverage` reports the mailboxes that are NOT protected rather than omitting them, so the number is a coverage statement an admin can act on. `connection test` takes the `mailsec_provider` RECORD NAME, never a credential. +`--topic` and `--subscription` on `onboarding` override the suggested Workspace +Pub/Sub names. Workspace needs domain-wide delegation and a topic and pull +subscription in the service account's own Google Cloud project. Microsoft 365 uses +an Entra application with admin-consented application permissions. Follow the +returned scopes, including optional capabilities, before saving a connection. + +For example, after creating the Microsoft application, save a secret file whose +`secret` value is the serialized credential JSON: + +```json +{"secret":"{\"tenant_id\":\"YOUR_TENANT_ID\",\"client_id\":\"YOUR_CLIENT_ID\",\"client_secret\":\"YOUR_CLIENT_SECRET\"}"} +``` + +Save the connection body as `connection.json`: + +```json +{ + "provider": "m365", + "credentials": "hive://secret/m365-mail", + "scope": {"include_addresses": ["pilot@corp.example"]}, + "ingest": {"mode": "push", "backfill_days": 14} +} +``` + +```bash +limacharlie secret set --key m365-mail --input-file credential-secret.json --enabled +limacharlie hive set --hive-name mailsec_provider --key m365-mail --input-file connection.json --enabled +limacharlie mailsec connection test m365-mail +limacharlie mailsec coverage +limacharlie mailsec message list --mailbox pilot@corp.example +``` + +Hive records are disabled by default, so `--enabled` is essential. The example +protects one pilot mailbox; an empty include list covers all discovered mailboxes. +Workspace uses `provider: gworkspace`, a service-account credential with +`admin_email`, explicit `ingest.mode: push`, and `features.pubsub_topic` and +`features.pubsub_subscription` containing full resource names from its setup guide. +The default backfill is 14 days; explicit `backfill_days: 0` disables it. Backfill +judges historical mail without emitting live message events or performing automatic +remediation. New organizations have no automatic remediation configured; add an +automation policy deliberately after validating coverage and verdicts. + +An optional capability can be unavailable while the connection still works. Read +each diagnostic check's `required`, `status` and `remediation` fields. `--include-watch` +establishes a real Workspace watch. Coverage confirms ongoing protection after the +credential test; investigate any unprotected or error mailboxes it reports. + +Coverage defaults to a cached 24-hour volume window. Explicit windows are +recomputed and rate-limited. Use `--window-days` or `--since`/`--until`, never both; +`volume.truncated` means the requested period exceeds retained message history. + ## The triage queue Repeatable filters are OR within a key and AND across keys. Cursors are opaque and passed back verbatim; changing a filter mid-walk is an error, not a differently-meaning page. @@ -45,19 +112,36 @@ Repeatable filters are OR within a key and AND across keys. Cursors are opaque a limacharlie mailsec message list --verdict suspicious --verdict malicious limacharlie mailsec message list --mailbox cfo@corp.example --since 2026-08-01 limacharlie mailsec message list --user-reported +limacharlie mailsec message list --lane backfill --since 2026-09-01T00:00:00Z limacharlie mailsec message list --link-domain evil.example # IOC pivot limacharlie mailsec message list --attachment-sha256 limacharlie mailsec message list --search "invoice overdue" --since 2026-08-01 limacharlie mailsec message get # the drawer limacharlie mailsec message similar # who else got this limacharlie mailsec message revisions # verdict history +limacharlie mailsec message revisions --limit 1000 ``` -An unknown message id returns a null message rather than an error: the index has a 35-day TTL, so a miss is normal. +`--lane live` selects newly arriving mail and `--lane backfill` selects onboarding +history; omitting it includes either. Lane works with time, verdict and IOC queries. +Combining it with `--mailbox`, `--sender-email` or `--campaign-id` returns +`lane_unsupported`. Keep the lane and other filters unchanged when following cursors. + +An unknown message id returns a null message rather than an error. The message +index is retained for at most 35 days; the organization's retention policy can +shorten that period. The drawer normally serves the preserved MDM and original +enrichments (`mdm_source: stored`). Its raw-EML fallback (`eml_reparse`) omits +enrichments; expired content returns `mdm: null` and `mdm_unavailable_reason`. + +`message similar` returns bounded clustering-key neighbours, not necessarily the +same campaign, and has no pagination. Read each candidate's `matched_keys` and +the response's lookback windows. Use `message list --campaign-id` for a paginated +campaign membership query. Revision history also has no cursor: inspect +`revisions_truncated` before treating it as a complete audit export. ## Raw EML -A different privilege from opening the drawer, because it takes a person's actual mail out of the building. `--justification` is required and is written to the access audit with your identity. +A different privilege from opening the drawer, because it takes a person's actual mail out of the building. Downloading requires both `mailsec.get` and `mailsec.get.eml`. `--justification` is required and is written to the access audit with your identity. These are the sender's own bytes, unmodified, so the command **refuses to write them to a terminal**: a hostile message carrying ANSI escape sequences would repaint your screen. Give it `--out-file`, or pipe it. Redirects and pipes are unaffected; `--to-terminal` overrides the refusal if you really want the bytes on screen. The check happens before the download, so a refusal records no access. @@ -77,6 +161,14 @@ limacharlie mailsec message action --action quarantine_message --reas limacharlie mailsec message action --action restore_message ``` +Single-message actions also include `move_to_spam`, `banner_message`, +`unbanner_message`, `submit_to_triage` and `crawl_link`. Banners use the organization's +`banners` policy, with optional provider scopes required by Workspace. Triage +submission records an `EMAIL_ACTION` for a configured AI trigger to consume; +it does not itself start an agent session. Link crawling requests analysis and +can spend the organization's analysis budget. Inspect action results rather +than assuming an accepted request changed message placement. + Bulk remediation is two-step: without `--confirm` it previews, and the preview's `confirm` token is derived from the normalized selection, so it can only execute what you previewed. Up to 500 messages per call — a larger selection is refused, not truncated. ```bash @@ -114,6 +206,26 @@ limacharlie mailsec rule backtest --file rule.json --since 2026-08-01 `rule backtest` reports `precision: null` — not `0` — when nothing it matched has an analyst disposition yet, and counts what it could not examine, so a precision figure whose denominator silently shrank is visible as one. +Validation and backtesting do not save a rule. Save an accepted rule with +`hive set --hive-name dr-mail --key --input-file rule.json --enabled`. +An invalid rule returns `valid: false` with its reason; check that field even +when the request succeeds. To disable a vendor rule, set its metadata disabled +instead of deleting it: vendor pack releases can recreate deleted vendor rules. + +Additional extension workflows use the generic CLI (requires `ext.request`): + +```bash +limacharlie extension request --name ext-email-security --action restore_default_rules +limacharlie extension request --name ext-email-security --action get_dlp_pack +``` + +`restore_default_rules` creates missing defaults without overwriting existing +records. `get_dlp_pack` only returns the opt-in outbound DLP definitions; it does +not install them. Install the returned lookup records before their `dr-general` +rules and preserve each record's `usr_mtd.enabled` setting, or use the console's +installation flow. Outbound mail is observation only; DLP detections do not stop +delivery or remediate sent messages. + ## Tenant purge Permanently deletes everything Email Security holds for the org: the message index and the long-term evidence lane, campaigns, sender profiles, the remediation audit trail, user reports, the stored raw messages and their parsed copies, the link-detonation results, and the org's Email Security connection and policy configuration. It also stops the mail connections at Microsoft or Google, so the provider stops sending notifications. diff --git a/doc/cli/hive-data.md b/doc/cli/hive-data.md index 423e4398..403eaa67 100644 --- a/doc/cli/hive-data.md +++ b/doc/cli/hive-data.md @@ -16,6 +16,22 @@ limacharlie hive delete --hive-name secret --key my-key --confirm The flag is `--hive-name`; `--category` does not exist and never has. +### Security product configuration + +`limacharlie hive list-types` includes the product configuration hives: + +| Hive | Purpose | +|---|---| +| `cloudsec_provider` | Cloud, identity, SaaS, and source-control connections | +| `cloudsec_policy` | Cloud posture and Code Security policies | +| `cloudsec_query` | Saved Cloud Security graph queries | +| `cloudsec_code_rule` | Enabled Code Security rules, including editable defaults and custom rules | +| `mailsec_provider` | Microsoft 365 and Google Workspace mail connections | +| `mailsec_policy` | Email Security policies | +| `dr-mail` | Email detection and verdict rules | + +Store provider credentials in `secret` records and reference them with `hive://secret/`. New records default to disabled; use `--enabled` when ready. See the [Cloud Security](cloud-security.md) and [Email Security](email-security.md) onboarding guides before creating connections. + ### Expiry A record can expire. `usr_mtd.expiry` is a Unix epoch in **milliseconds** (`0` = never), diff --git a/doc/getting-started.md b/doc/getting-started.md index 45834c8c..d678d5f4 100644 --- a/doc/getting-started.md +++ b/doc/getting-started.md @@ -10,6 +10,18 @@ pip install limacharlie The `--output toon` format needs the optional `toon` extra (`pip install 'limacharlie[toon]'`); everything else works with the install above. See [Output Formats](cli/README.md#output-formats) for the uv caveat. +### Security products + +Install or upgrade the CLI with the normal package release: + +```bash +python -m pip install --upgrade limacharlie +limacharlie mailsec --help +limacharlie cloudsec code --help +``` + +Use the [Cloud Security](cli/cloud-security.md) and [Email Security](cli/email-security.md) references, or `limacharlie help cloud-security`, `limacharlie help code-security`, and `limacharlie help email-security`, for onboarding. + Docker: ```bash diff --git a/doc/sdk/README.md b/doc/sdk/README.md index 059d6ace..b6096297 100644 --- a/doc/sdk/README.md +++ b/doc/sdk/README.md @@ -60,6 +60,8 @@ org = Organization(client) | `AI` | `limacharlie.sdk.ai` | AI-assisted rule/query generation | | `Billing` | `limacharlie.sdk.billing` | Billing and usage details | | `Cases` | `limacharlie.sdk.cases` | SOC case management, investigation tracking, reporting | +| `CloudSec` | `limacharlie.sdk.cloudsec` | Cloud and Code Security: posture, findings, repositories, images, and remediation | +| `Mailsec` | `limacharlie.sdk.mailsec` | Email Security: onboarding, coverage, messages, verdicts, remediation, and reports | ## Raw API Requests @@ -106,6 +108,7 @@ status, data = client.raw_request("GET", f"orgs/{client.oid}", | [Search & Insight](search-insight.md) | LCQL queries, IOC search, enrichment | | [Streaming](streaming.md) | Spout, Firehose | | [Configuration Sync](configs.md) | Infrastructure-as-code | +| [Security Products](security-products.md) | Cloud Security, Code Security, Email Security, setup, and pagination | | [Other Classes](other-classes.md) | Extensions, Artifacts, Payloads, Outputs, AI, Billing, Cases | ## See Also diff --git a/doc/sdk/other-classes.md b/doc/sdk/other-classes.md index 15d37bbd..31abdd91 100644 --- a/doc/sdk/other-classes.md +++ b/doc/sdk/other-classes.md @@ -188,7 +188,11 @@ cs.test_provider({"provider_type": "gcp", "credentials": "hive://secret/gcp-sa"} cs.get_provider_manifests(provider_type="gcp") ``` -The Cloud Security (CNAPP) surface: findings (CSPM + attack paths + CIEM) with their facet and shared-fix rollups, resource inventory, security graph queries, compliance, the identity access population, DSPM stores and facets, CAASM ingest/coverage, sensor↔asset resolution, free-tier standing, fleet overview, and CSV exports. See [CLI: cloudsec](../cli/cloud-security.md) for the command-line equivalents. +The Cloud Security (CNAPP) surface also includes Code Security: repository and container-image scans, build provenance, sanitized IaC mapping, code-to-cloud impact, runtime evidence, and governed remediation. Availability depends on the provider and deployment. See [Security Products](security-products.md) for SDK onboarding and [CLI: cloudsec](../cli/cloud-security.md) for the command-line equivalents. + +## Mailsec + +`limacharlie.sdk.mailsec.Mailsec` wraps Email Security onboarding, provider tests, coverage, messages, justified EML downloads, verdict revisions, campaigns, remediation, and user reports. See [Security Products](security-products.md#email-security) for setup, examples, and pagination, and [CLI: mailsec](../cli/email-security.md) for complete workflows. ## See Also diff --git a/doc/sdk/security-products.md b/doc/sdk/security-products.md new file mode 100644 index 00000000..dbc39c3c --- /dev/null +++ b/doc/sdk/security-products.md @@ -0,0 +1,94 @@ +[Documentation](../README.md) > [SDK](README.md) > Security Products + +# Cloud Security, Code Security, and Email Security + +Use `CloudSec` for both Cloud Security and Code Security, and `Mailsec` for Email Security. Both take an authenticated `Organization`; its UUID determines which tenant every request addresses. + +Install or upgrade with `python -m pip install --upgrade limacharlie`. See [installation](../getting-started.md#installation) for setup. + +```python +from limacharlie.client import Client +from limacharlie.sdk.organization import Organization +from limacharlie.sdk.cloudsec import CloudSec +from limacharlie.sdk.mailsec import Mailsec + +# Uses the selected organization's saved CLI credentials. +org = Organization(Client()) +cloud = CloudSec(org) +mail = Mailsec(org) +``` + +Enable each product once using `limacharlie extension subscribe --name ext-cloud-security` or `--name ext-email-security`. Save credentials in the secret Hive and reference them from `cloudsec_provider` or `mailsec_provider` records. Follow the full [Cloud Security](https://docs.limacharlie.io/cloud-security/getting-started/) and [Email Security](https://docs.limacharlie.io/email-security/getting-started/) provider guides. + +## Cloud and Code Security + +Check collection health and coverage before reviewing findings. Code scanning additionally needs a supported source-control connection and a `cloudsec_policy` record of type `code_scanning`. + +```python +print(cloud.get_scan_status()) +print(cloud.get_provider_manifests(provider_type="gcp")) +print(cloud.list_findings(severity=["CRITICAL", "HIGH"], status=["open"])) + +print(cloud.get_code_capabilities()) +print(cloud.get_code_status()) +for repository in cloud.iter_code_repos(limit=100): + print(repository["repo"], repository.get("scan_status")) +print(cloud.get_code_coverage()) +``` + +`partial`, `unknown`, stale evidence, or an unmatched code-to-cloud join describes a coverage gap. Scans are asynchronous: acceptance is not completion. Availability depends on the provider and deployment; consult `get_code_capabilities()` before configuring PR checks or remediation. + +`cloudsec.get` permits ordinary reads. `cloudsec.set` permits ordinary writes; AutoFix and remediation decisions require `cloudsec.respond`. Provider records use `cloudsec_provider.*` and credentials use `secret.*`; policy, saved-query, and code-rule Hives reuse `cloudsec.get/set`. + +See the [Cloud and Code Security CLI reference](../cli/cloud-security.md) for build provenance, sanitized IaC maps, runtime evidence, container lineage, and governed remediation. + +## Email Security + +Connect a small pilot scope first. The connection test accepts the **saved provider record name**, rather than a credential or provider name. + +```python +print(mail.get_onboarding(provider="m365")) +print(mail.test_connection("pilot")) +print(mail.get_coverage(window_days=7)) +print(mail.list_messages(limit=20)) # confirm benign pilot mail arrived, too + +# Keep the same filters while walking opaque cursors. +cursor = None +while True: + page = mail.list_messages( + verdict=["suspicious", "malicious"], cursor=cursor, limit=200, + ) + for message in page.get("messages", []): + print(message) + cursor = page.get("next_cursor") + if not cursor: + break +``` + +`get_message(msg_uuid)` returns the indexed message and its parsed representation, preferring the preserved message used to judge it. Inspect `mdm_source` and `mdm_unavailable_reason`: older or unavailable content may use a reparse or have no parsed representation. + +Raw EML is bytes and requires a separate permission and an audited justification: + +```python +raw = mail.get_message_eml("", justification="incident investigation") +with open("suspect.eml", "wb") as output: + output.write(raw) +``` + +To inspect a local file without ingesting it, preserve its bytes with base64: + +```python +import base64 + +with open("suspect.eml", "rb") as source: + encoded = base64.b64encode(source.read()).decode("ascii") +print(mail.analyze(eml_b64=encoded, org_domains=["corp.example"])) +``` + +`mailsec.get` permits structured reads, `mailsec.set` changes triage and rules, and `mailsec.act` remediates provider mail, revises verdicts, and tests connections. Original-byte downloads require both `mailsec.get` and `mailsec.get.eml`. Provider records use `mailsec_provider.*` and credentials use `secret.*`; policy and `dr-mail` Hives reuse `mailsec.get/set`. + +Start with `alert_only`. Manual provider actions need an explicit `force=True` override in that mode; inspect the action and audit outcome. Bulk and campaign actions use preview and confirmation so the executed selection matches what was reviewed. See the [Email Security CLI reference](../cli/email-security.md) for these workflows, verdict revisions, campaigns, user reports, and offboarding. + +## Pagination and filters + +List methods return one page unless documented as iterators. A short page can still have a `next_cursor`; continue until it is empty. Keep the original filters and return cursors verbatim. Boolean selectors are tri-state: omit them for no constraint, use `True` for a positive selection, and `False` for a negative selection. `list_similar_messages()` returns a bounded candidate set and does not support cursor pagination. diff --git a/limacharlie/commands/cloudsec.py b/limacharlie/commands/cloudsec.py index b7d9c254..8c55fcf5 100644 --- a/limacharlie/commands/cloudsec.py +++ b/limacharlie/commands/cloudsec.py @@ -65,7 +65,7 @@ # compiled into the binary) or `--rules-file` (a rule set document). Older scanners reject both # flags, which is why MIN_RULES_SCANNER_VERSION is named in the error a custom --image/--binary # gets when it does. -DEFAULT_CODE_SCANNER_IMAGE = "gcr.io/legion-212720/github.com/refractionpoint/lc-code-scanner:v0.16.0" +DEFAULT_CODE_SCANNER_IMAGE = "gcr.io/legion-212720/github.com/refractionpoint/lc-code-scanner:v0.24.0" MIN_RULES_SCANNER_VERSION = "v0.16.0" # The hive an organization's static-analysis rules live in, one Opengrep rule file per record, @@ -744,7 +744,10 @@ def _stamp_sarif_execution(document: bytes, succeeded: bool) -> tuple[bytes, int The merged third-party asset inventory: every device/identity the org's connected tools (EDR / IdP / MDM / scanners) report, entity-resolved to one row per real asset with per-source -provenance in props. +provenance in props. Filter with repeatable --kind, --source and +--posture-encryption/--posture-screen-lock/--posture-compromised/--posture-managed. +An empty posture value selects unreported facts, not compliant assets. +--sort last_seen shows newest observations first; urn is the stable default. Examples: limacharlie cloudsec caasm assets @@ -845,7 +848,9 @@ def _stamp_sarif_execution(document: bytes, succeeded: bool) -> tuple[bytes, int Export the (filtered) findings worklist as CSV. The server walks the FULL filtered set (no pagination), capped at 100k rows — a trailing '#' comment row marks a truncated export. Takes the same filters as -'finding list'. +'finding list'. Use --max-rows to bound each request and resume with +--cursor from the trailing '# next_cursor=' row. Keep filters unchanged. +Without --max-rows the export starts at the beginning. Examples: limacharlie cloudsec export findings -o findings.csv @@ -855,7 +860,9 @@ def _stamp_sarif_execution(document: bytes, succeeded: bool) -> tuple[bytes, int _EXPLAIN_EXPORT_INVENTORY = """\ Export the (filtered) cloud resource inventory as CSV. The server walks the full filtered set (no pagination), capped at 100k rows. -Takes the same filters as 'inventory list'. +Takes the same filters as 'inventory list'. Use --max-rows to bound each request +and resume with --cursor from its trailing '# next_cursor=' row. +Keep filters unchanged; --cursor requires --max-rows. Examples: limacharlie cloudsec export inventory -o inventory.csv @@ -1053,6 +1060,14 @@ def _stamp_sarif_execution(document: bytes, succeeded: bool) -> tuple[bytes, int register_explain("cloudsec.caasm.policy.get", _EXPLAIN_CAASM_POLICY_GET) register_explain("cloudsec.caasm.policy.set", _EXPLAIN_CAASM_POLICY_SET) register_explain("cloudsec.caasm.ingest", _EXPLAIN_CAASM_INGEST) +register_explain("cloudsec.provider.m365-certificate", """Generate an Entra/Microsoft 365 connection certificate; requires cloudsec.set +and secret.set. The private key stays in the organization secret store. +Repeat calls return the existing certificate. --replace updates the stored key pair +immediately; existing authentication may stop until the new certificate is uploaded. +Use --out connection.cer to save the public certificate, upload it to your Entra +app registration, and put the returned credentials reference in cloudsec_provider. +Example: limacharlie cloudsec provider m365-certificate my-entra --out connection.cer +""") register_explain("cloudsec.provider.test", _EXPLAIN_PROVIDER_TEST) register_explain("cloudsec.provider.manifest", _EXPLAIN_PROVIDER_MANIFEST) register_explain("cloudsec.policy.vocabulary", _EXPLAIN_POLICY_VOCABULARY) @@ -1843,16 +1858,16 @@ def code_status(ctx) -> None: @click.option("--repo", default=None, help="Narrow to the connection covering one repository " "('/' as 'code repos' returns it). Omit " - "to list every GitHub connection.") + "to list enabled workflow connections.") @pass_context def code_capabilities(ctx, repo) -> None: """What each source-control connection may actually DO: scanning, PR checks, PR comments, dependency AutoFix. - Covers GitHub connections only — a GitLab or Bitbucket connection scans - with its own read-only token and has no write plane to detect, so it - never appears here (not even as 'unknown'); 'cloudsec provider - manifest' is the coverage view for those. + GitHub connections are always reported. GitLab.com and Bitbucket Cloud + connections appear when their workflow support is enabled in this + deployment; absence does not mean repository scanning is off. Use + 'cloudsec provider manifest' to inspect collection coverage. A capability of 'available' means the control MAY be offered, not that it fires on its own: 'pr_checks' reading 'available' says the @@ -1985,13 +2000,8 @@ def code_rescan(ctx, repo, ref, provider) -> None: @click.argument("repo") @click.option("--pr", required=True, type=int, help="The pull-request number.") -@click.option("--base-sha", required=True, - help="A FULL commit id (40 or 64 hex characters). Still " - "required, but NOT authoritative: the lane takes the base " - "from the provider, because a caller-chosen base decides " - "what the diff is measured from and a base equal to the " - "head would make any pull request look like it introduced " - "nothing.") +@click.option("--base-sha", default=None, + help="Full base commit id. Required for GitHub; optional for GitLab.com and Bitbucket Cloud. The provider resolves the authoritative base.") @click.option("--head-sha", required=True, help="The FULL commit id of the head. A branch or tag name is " "refused: the check is published ON the commit, and a ref " @@ -2023,8 +2033,8 @@ def code_pr_check(ctx, repo, pr, base_sha, head_sha, action, prev_base_sha, REPO is '/', the bare repository name, or its urn. - The lane scans the pull request's base and head and publishes a GitHub - check run on the head commit reporting only what is NEW in it; the + The lane scans the pull request's base and head and publishes a provider + check on the head commit reporting only what is NEW in it; the repository's own findings stay on 'cloudsec code repos'. Its normal caller is the shipped D&R rule on the org's source-control webhook — this is the same door for a CI job. @@ -2060,6 +2070,11 @@ def code_pr_check(ctx, repo, pr, base_sha, head_sha, action, prev_base_sha, limacharlie cloudsec code pr-check acme/api --pr 42 --action edited \\ --base-sha --head-sha --prev-base-sha """ + selected_provider = (provider or "github").strip().lower() + if selected_provider not in ("gitlab", "bitbucket") and not base_sha: + raise click.UsageError("--base-sha is required for GitHub pull-request checks") + if selected_provider in ("gitlab", "bitbucket") and action == "edited": + raise click.UsageError("--action edited is GitHub-only") if action == "edited" and not prev_base_sha: raise click.UsageError( "--action edited needs --prev-base-sha: an edited pull request is " @@ -2174,17 +2189,13 @@ def code_autofix(ctx, finding_id, repo, provider) -> None: with no recorded deployment scope ends 'pr_merged_unverifiable'; a PR closed without merge ends 'pr_closed'. Neither means verified. - Lockfiles: for npm the package-lock.json IS rewritten by default. One - read-only registry metadata document supplies the new version's resolved - URL and integrity digest; it is left stale only where the code_scanning - policy sets 'autofix_registry_access: false', where the lock is a - yarn.lock or pnpm-lock.yaml, or where the entry could not be rewritten - safely. For go the go.sum is NOT regenerated — it hashes a module zip - nobody downloaded — and only matters where the tree has one. pip - (requirements.txt) and maven have no lockfile, so those changes are - complete. Wherever a lock is left stale the pull request says so - prominently and names the command to run; trust the pull request over - this summary. + Lockfiles: npm lockfiles (package-lock.json, npm-shrinkwrap.json, + yarn.lock and pnpm-lock.yaml) are updated when supported. An unsupported + yarn or pnpm rewrite is refused before a job runs. A package-lock that + cannot be updated (for example with autofix_registry_access false) is + flagged lockfile_stale with the command to run. For go, go.sum is updated + from verified checksum-database entries; a missing or uncompletable go.sum + refuses the fix. Review the PR's recorded outcome and limitations. \b Examples: @@ -2403,7 +2414,7 @@ def code_coverage(ctx, summary) -> None: @click.option("--ingest/--no-ingest", "do_ingest", default=False, help="Push the report to LimaCharlie when the scan finishes.") @click.option("--image", "image", default=None, - help="Scanner image to run (default: the published lc-code-scanner).") + help="Scanner image to run. The default requires registry access; use an image you can pull or --binary.") @click.option("--binary", "binary", default=None, help="Run this scanner-agent binary instead of the container.") @click.option("-o", "--output", "output_path", default=None, @@ -2445,7 +2456,9 @@ def code_scan(ctx, path, repo, commit, do_ingest, image, binary, nothing is duplicated. The scanner runs in a container by default; --binary runs an already - installed scanner-agent instead. Nothing about the checkout leaves your + installed scanner-agent instead. The default image requires registry + access; supply --image with an accessible scanner image or --binary + if you cannot pull it. Nothing about the checkout leaves your machine except the report. STATIC ANALYSIS RULES. The scanner has no rules of its own. When sast is @@ -2899,6 +2912,11 @@ def _run(cmd: list[str], timeout_s: int, *, env: dict | None = None, pass raise click.ClickException( "the scan did not finish within %d seconds" % timeout_s) + if container and proc.returncode in (125, 126, 127): + raise click.ClickException( + "Docker could not start the scanner (exit %d). Check the Docker error " + "and registry access, or use --image with an accessible scanner image " + "or --binary with an installed scanner-agent." % proc.returncode) if proc.returncode != 0: # The agent's exit codes are a closed vocabulary and it prints a # machine-readable line before a fatal exit, so the caller is pointed at @@ -3035,10 +3053,12 @@ def image_repos(ctx, q, providers, accounts, registries, regions, @image_group.command("repo-facets") +@click.option("--lineage-facet/--no-lineage-facet", default=None, + help="Include digest-global effective lineage counts. Repository filters do not narrow them.") @_image_repo_filter_options @pass_context def image_repo_facets(ctx, q, providers, accounts, registries, regions, - has_findings, has_images, scanning_state) -> None: + has_findings, has_images, scanning_state, lineage_facet) -> None: """Cross-filtered facet counts for the image-repository list. Takes the same selectors as 'image repos' (this endpoint has no @@ -3066,10 +3086,14 @@ def image_repo_facets(ctx, q, providers, accounts, registries, regions, has_findings=has_findings, has_images=has_images, scanning_state=scanning_state, + lineage_facet=lineage_facet, )) @image_group.command("list") +@click.option("--lineage-status", "lineage_statuses", multiple=True, + type=click.Choice(["verified", "asserted", "inferred", "ambiguous", "unknown"]), + help="Effective source-lineage status; repeatable (OR). Stale decisions match unknown.") @click.option("-q", "--search", "q", default=None, help="Substring over the image name and urn, and the joined " "repository path, registry host and tags.") @@ -3109,7 +3133,7 @@ def image_repo_facets(ctx, q, providers, accounts, registries, regions, @_paging_options @pass_context def image_list(ctx, q, repo_urns, providers, accounts, registries, tags, - findings, running, signed, sort, order, walk_all, cursor, + findings, running, signed, lineage_statuses, sort, order, walk_all, cursor, limit) -> None: """List container images, keyed by digest. @@ -3142,6 +3166,7 @@ def image_list(ctx, q, repo_urns, providers, accounts, registries, tags, findings=findings, running=running, signed=signed, + lineage_status=list(lineage_statuses) or None, sort=sort, order=order, ) @@ -4565,10 +4590,19 @@ def caasm_group() -> None: @caasm_group.command("assets") +@click.option("--kind", "kinds", multiple=True, help="Asset kind, for example device or user; repeatable (OR).") +@click.option("--source", "sources", multiple=True, help="Observing tool; repeatable (OR).") +@click.option("--posture-encryption", multiple=True, help="Reported encryption value; repeatable. Empty selects unreported.") +@click.option("--posture-screen-lock", multiple=True, help="Reported screen-lock value; repeatable. Empty selects unreported.") +@click.option("--posture-compromised", multiple=True, help="Reported compromised value; repeatable. Empty selects unreported.") +@click.option("--posture-managed", multiple=True, help="Reported managed value; repeatable. Empty selects unreported.") +@click.option("--sort", default=None, type=click.Choice(["urn", "last_seen"]), + help="Stable urn order (default) or newest observation first.") @click.option("-q", "--search", "q", default=None, help="Substring filter over asset urn/name.") @_paging_options @pass_context -def caasm_assets(ctx, q, cursor, limit) -> None: +def caasm_assets(ctx, q, kinds, sources, posture_encryption, posture_screen_lock, + posture_compromised, posture_managed, sort, cursor, limit) -> None: """The merged third-party asset inventory. \b @@ -4576,7 +4610,14 @@ def caasm_assets(ctx, q, cursor, limit) -> None: limacharlie cloudsec caasm assets -q laptop --limit 50 """ cs = _get_cloudsec(ctx) - _output(ctx, cs.list_caasm_assets(q=q, cursor=cursor, limit=limit)) + _output(ctx, cs.list_caasm_assets( + q=q, kind=list(kinds) or None, source=list(sources) or None, + posture_encryption=list(posture_encryption) or None, + posture_screen_lock=list(posture_screen_lock) or None, + posture_compromised=list(posture_compromised) or None, + posture_managed=list(posture_managed) or None, sort=sort, + cursor=cursor, limit=limit, + )) @caasm_group.command("coverage") @@ -4713,6 +4754,44 @@ def provider_manifest(ctx, provider_type) -> None: _output(ctx, cs.get_provider_manifests(provider_type=provider_type)) +@provider_group.command("m365-certificate") +@click.argument("connection") +@click.option("--client-id", default=None, help="Entra application's client id (GUID).") +@click.option("--replace", is_flag=True, default=False, + help="Replace the stored key pair immediately. Existing authentication may stop until you upload the new public certificate.") +@click.option("--out", "output_path", default=None, type=click.Path(dir_okay=False), + help="Save the public DER certificate as a .cer file for Entra upload.") +@pass_context +def provider_m365_certificate(ctx, connection, client_id, replace, output_path) -> None: + """Generate an Entra/Microsoft 365 connection certificate. + + Requires cloudsec.set and secret.set. The private key stays in the + organization's secret store. Repeat calls return the existing certificate. + Upload the public certificate to your Entra app registration and use the + returned credentials reference in your cloudsec_provider record. + + \b + Example: + limacharlie cloudsec provider m365-certificate my-entra --out connection.cer + """ + result = _get_cloudsec(ctx).mint_m365_certificate( + connection, client_id=client_id, replace=replace, + ) + if output_path: + try: + certificate = base64.b64decode(result["certificate"], validate=True) + if not certificate: + raise ValueError("empty certificate") + except (KeyError, TypeError, ValueError): + raise click.ClickException("server did not return a valid public certificate") from None + try: + with open(output_path, "wb") as destination: + destination.write(certificate) + except OSError as error: + raise click.ClickException("cannot write public certificate: %s" % error) from error + _output(ctx, result) + + @provider_group.command("test") @click.option("--provider-json", default=None, help="The provider record (cloudsec_provider hive shape) as inline JSON.") @@ -4893,6 +4972,13 @@ def _emit_csv(ctx: click.Context, csv_text: str, output_path: str | None) -> Non click.echo(csv_text, nl=False) +def _export_chunk_options(f): + f = click.option("--max-rows", default=None, type=click.IntRange(1, 100000), + help="Bound findings/inventory CSV to roughly this many rows, rounded to 1000-row pages. Read the trailing next_cursor comment.")(f) + return click.option("--cursor", default=None, + help="Resume token from a bounded export; requires --max-rows and unchanged filters.")(f) + + def _export_output_option(f): return click.option( "-o", "--output-file", "output_path", default=None, @@ -4911,6 +4997,7 @@ def export_group() -> None: @export_group.command("findings") +@_export_chunk_options @_finding_filter_options @click.option("--cause", default=None, help="Exact shared-fix cause key, as 'cloudsec finding causes' " @@ -4922,7 +5009,7 @@ def export_group() -> None: def export_findings(ctx, has_iac_origin, iac_attributions, severities, finding_classes, statuses, accounts, repos, image_urns, fix_states, exploit_bands, grains, source, owners, unassigned, sla_states, reachable, kev, q, - cause, sort, order, output_path) -> None: + cause, sort, order, output_path, max_rows, cursor) -> None: """Export the (filtered) findings worklist as CSV. \b @@ -4931,6 +5018,8 @@ def export_findings(ctx, has_iac_origin, iac_attributions, severities, finding_c limacharlie cloudsec export findings --severity CRITICAL --status open limacharlie cloudsec export findings --owner alice@corp.com """ + if cursor and max_rows is None: + raise click.UsageError("--cursor requires --max-rows for a resumable export") cs = _get_cloudsec(ctx) _emit_csv(ctx, cs.export_findings_csv( has_iac_origin=has_iac_origin, @@ -4953,15 +5042,17 @@ def export_findings(ctx, has_iac_origin, iac_attributions, severities, finding_c q=q, sort=sort, order=order, + max_rows=max_rows, cursor=cursor, ), output_path) @export_group.command("inventory") +@_export_chunk_options @_inventory_filter_options @_export_output_option @pass_context def export_inventory(ctx, has_iac_origin, resource_type, provider, account, region, q, - all_accounts, account_empty, output_path) -> None: + all_accounts, account_empty, output_path, max_rows, cursor) -> None: """Export the (filtered) cloud resource inventory as CSV. \b @@ -4969,10 +5060,13 @@ def export_inventory(ctx, has_iac_origin, resource_type, provider, account, regi limacharlie cloudsec export inventory -o inventory.csv limacharlie cloudsec export inventory --provider okta """ + if cursor and max_rows is None: + raise click.UsageError("--cursor requires --max-rows for a resumable export") cs = _get_cloudsec(ctx) _emit_csv(ctx, cs.export_inventory_csv( resource_type=resource_type, provider=provider, account=account, region=region, q=q, has_iac_origin=has_iac_origin, + max_rows=max_rows, cursor=cursor, account_empty=_inventory_account_empty( account, all_accounts, account_empty), ), output_path) diff --git a/limacharlie/commands/hive.py b/limacharlie/commands/hive.py index ee930d8f..4bc74bd9 100644 --- a/limacharlie/commands/hive.py +++ b/limacharlie/commands/hive.py @@ -102,6 +102,10 @@ def _record_from_input(key: str, data: Any) -> HiveRecord: # Known hive types supported by LimaCharlie. _KNOWN_HIVE_TYPES = [ + "cloudsec_provider", + "cloudsec_policy", + "cloudsec_query", + "cloudsec_code_rule", "mailsec_provider", "mailsec_policy", "dr-mail", @@ -714,7 +718,9 @@ def rename(ctx, hive_name, key, new_name) -> None: Hives are key-value stores that hold different types of configuration data for a LimaCharlie organization. Each hive type stores a specific -kind of data. Known types: dr-general, dr-managed, dr-service, fp, +kind of data. Cloud Security uses cloudsec_provider, cloudsec_policy, +cloudsec_query, and cloudsec_code_rule. Email Security uses mailsec_provider, +mailsec_policy, and dr-mail. Other known types: dr-general, dr-managed, dr-service, fp, cloud_sensor, extension_config, yara, lookup, secret, query, playbook, ai_agent, ai_skill, ai_memory, external_adapter, sop, org_notes. diff --git a/limacharlie/commands/mailsec.py b/limacharlie/commands/mailsec.py index 10d9df13..ac5a2664 100644 --- a/limacharlie/commands/mailsec.py +++ b/limacharlie/commands/mailsec.py @@ -64,6 +64,11 @@ so the number is a coverage statement an admin can act on instead of a count of whatever happened to work. +Use --since/--until for a specific range, or --window-days for days +back from now. These forms cannot be combined. Explicit windows are +recomputed and rate-limited; the default 24-hour window is cached. +volume.truncated means the requested period exceeds retained messages. + Examples: limacharlie mailsec coverage limacharlie mailsec coverage --window-days 30 @@ -73,6 +78,11 @@ The message index — the triage queue. Repeatable filters OR within a key and AND across keys. +--lane live selects newly arriving mail; --lane backfill selects history +judged during onboarding. Omit it for either. The lane filter works with +time, verdict and IOC queries, but the server refuses it alongside +--mailbox, --sender-email or --campaign-id (lane_unsupported). + --user-reported is worth knowing about: a human taking the trouble to report a message is the strongest signal this product gets, and it outranks whatever the scorer decided. @@ -85,17 +95,15 @@ """ _EXPLAIN_MESSAGE_GET = """\ -One message: the index row plus the re-parsed MDM (the drawer). +One message: the index row plus the parsed Message Data Model (MDM). -The MDM is re-parsed from the stored raw copy rather than read from -the index, and the response says which path produced it. Enrichments -are deliberately absent rather than recomputed — they were resolved -against sender profiles as they existed at ingest, and synthesising -today's values would show you a reputation the verdict was never -based on. +mdm_source=stored serves the preserved MDM the engine judged, with its +original enrichments. eml_reparse falls back to parsing the retained raw +copy with today's parser and leaves enrichments absent. Expired content +returns mdm:null with mdm_unavailable_reason; the index row can still exist. -An unknown id returns a null message, not an error: the index has a -35-day TTL, so a miss is normal. +An unknown id returns a null message, not an error. The message index is +retained for at most 35 days; your retention policy can shorten that window. Examples: limacharlie mailsec message get 0057db2b-3a06-5aab-b3be-c1e6c15dcf10 @@ -115,8 +123,9 @@ """ _EXPLAIN_MESSAGE_SIMILAR = """\ -Messages clustered with this one — "who else got this", answered from -a single message. +Recent messages sharing a clustering key with this one. These are +candidates, not necessarily members of the same campaign. Each row +names its matching keys, and the response states the lookback window. Examples: limacharlie mailsec message similar 0057db2b-... @@ -173,6 +182,9 @@ the rationale they gave — the audit of how a message's disposition moved over time. +--limit bounds the history returned. Check revisions_truncated before +treating the result as a complete audit export; this route has no cursor. + Examples: limacharlie mailsec message revisions 0057db2b-... """ @@ -484,16 +496,15 @@ """ _EXPLAIN_ONBOARDING = """\ -The setup guide for connecting a mail provider, with this org's own -values already substituted in. - -Served by the backend rather than written into the docs so the -identifiers you must paste - the service account, the topic, the -subscription - are the real ones for this deployment rather than -placeholders you have to translate. +Fetch current provider scopes and setup steps. Workspace uses resources +in your own Google Cloud project. Supply --project-id and --sa-email to +fill those values into its setup commands; --topic and --subscription +override the suggested names. Without substitutions the guide contains +placeholders. Fetching it creates no resources. Examples: limacharlie mailsec onboarding --provider gworkspace + limacharlie mailsec onboarding --provider gworkspace --project-id your-project --sa-email mailsec@your-project.iam.gserviceaccount.com limacharlie mailsec onboarding --provider m365 """ @@ -969,15 +980,24 @@ def tenant_group() -> None: @group.command("coverage") @click.option("--window-days", default=None, type=int, help="Days of volume to summarise.") +@click.option("--since", default=None, help="Start of the volume window (RFC3339 or unix seconds).") +@click.option("--until", default=None, help="End of the volume window (RFC3339 or unix seconds).") @pass_context -def coverage(ctx, window_days) -> None: +def coverage(ctx, window_days, since, until) -> None: """Mailbox coverage and analysed volume. \b Example: limacharlie mailsec coverage --window-days 30 """ - _output(ctx, _get_mailsec(ctx).get_coverage(window_days=window_days)) + if window_days is not None and (since is not None or until is not None): + raise click.UsageError("--window-days cannot be combined with --since or --until") + kwargs = {"window_days": window_days} + if since is not None: + kwargs["since"] = since + if until is not None: + kwargs["until"] = until + _output(ctx, _get_mailsec(ctx).get_coverage(**kwargs)) @group.command("analyze") @@ -1011,15 +1031,24 @@ def analyze(ctx, eml_file, org_domains, direction) -> None: @group.command("onboarding") @click.option("--provider", default=None, type=click.Choice(["m365", "gworkspace"]), help="Which provider's guide to fetch.") +@click.option("--project-id", default=None, help="Your Google Cloud project for Workspace setup commands.") +@click.option("--sa-email", default=None, help="Your Workspace service account email for setup commands.") +@click.option("--topic", default=None, help="Override the suggested Workspace Pub/Sub topic name.") +@click.option("--subscription", default=None, help="Override the suggested Workspace pull subscription name.") @pass_context -def onboarding(ctx, provider) -> None: - """Provider setup guide, with this org's own values filled in. +def onboarding(ctx, provider, project_id, sa_email, topic, subscription) -> None: + """Provider setup guide; supply Workspace values for copyable commands. \b Example: limacharlie mailsec onboarding --provider gworkspace """ - _output(ctx, _get_mailsec(ctx).get_onboarding(provider=provider)) + kwargs = {"provider": provider} + for key, value in (("project_id", project_id), ("sa_email", sa_email), + ("topic", topic), ("subscription", subscription)): + if value is not None: + kwargs[key] = value + _output(ctx, _get_mailsec(ctx).get_onboarding(**kwargs)) # --------------------------------------------------------------------------- @@ -1034,6 +1063,8 @@ def onboarding(ctx, provider) -> None: @click.option("--campaign-id", default=None, help="Only members of this campaign.") @click.option("--state", multiple=True, help="Message state (repeatable).") @click.option("--direction", multiple=True, help="inbound|outbound|internal (repeatable).") +@click.option("--lane", default=None, type=click.Choice(["live", "backfill"]), + help="Processing lane. Cannot be combined with --mailbox, --sender-email or --campaign-id.") @click.option("--user-reported", is_flag=True, default=False, help="Only mail a person reported.") @click.option("--no-user-reported", is_flag=True, default=False, help="Only mail nobody reported.") @click.option("--min-score", default=None, type=int, help="Only messages at or above this score.") @@ -1047,7 +1078,7 @@ def onboarding(ctx, provider) -> None: @click.option("--limit", default=None, type=int, help="Page size.") @pass_context def message_list(ctx, verdict, mailbox, sender_email, sender_domain, campaign_id, state, - direction, user_reported, no_user_reported, min_score, link_domain, + direction, lane, user_reported, no_user_reported, min_score, link_domain, attachment_sha256, q, since, until, cursor, limit) -> None: """The message index — the triage queue. @@ -1060,6 +1091,7 @@ def message_list(ctx, verdict, mailbox, sender_email, sender_domain, campaign_id """ ms = _get_mailsec(ctx) try: + lane_params = {"lane": lane} if lane is not None else {} result = ms.list_messages( verdict=list(verdict) or None, mailbox=mailbox, @@ -1068,6 +1100,7 @@ def message_list(ctx, verdict, mailbox, sender_email, sender_domain, campaign_id campaign_id=campaign_id, state=list(state) or None, direction=list(direction) or None, + **lane_params, user_reported=_tri_state(user_reported, no_user_reported, "user-reported"), min_score=min_score, link_domain=link_domain, @@ -1087,7 +1120,7 @@ def message_list(ctx, verdict, mailbox, sender_email, sender_domain, campaign_id @click.argument("msg_uuid") @pass_context def message_get(ctx, msg_uuid) -> None: - """One message: the index row plus the re-parsed MDM. + """One message: the index row plus the preserved or re-parsed MDM. \b Example: @@ -1148,23 +1181,29 @@ def message_eml(ctx, msg_uuid, justification, out_path, to_terminal) -> None: @message_group.command("similar") @click.argument("msg_uuid") -@click.option("--cursor", default=None, help="Keyset token from a previous page.") -@click.option("--limit", default=None, type=int, help="Page size.") +@click.option("--cursor", default=None, hidden=True) +@click.option("--limit", default=None, type=int, hidden=True) @pass_context def message_similar(ctx, msg_uuid, cursor, limit) -> None: - """Messages clustered with this one — who else got it. + """Recent messages sharing clustering keys; candidates for investigation. \b Example: limacharlie mailsec message similar 0057db2b-... """ - _output(ctx, _get_mailsec(ctx).list_similar_messages(msg_uuid, cursor=cursor, limit=limit)) + if cursor is not None or limit is not None: + raise click.UsageError( + "similar messages are not paginated; omit --cursor and --limit, " + "or use message list with --campaign-id or time/IOC filters" + ) + _output(ctx, _get_mailsec(ctx).list_similar_messages(msg_uuid)) @message_group.command("action") @click.argument("msg_uuid") @click.option("--action", "action_name", required=True, - help="quarantine_message|trash_message|restore_message|banner_message|unbanner_message") + help="quarantine_message|trash_message|move_to_spam|restore_message|banner_message|" + "unbanner_message|submit_to_triage|crawl_link") @click.option("--reason", default=None, help="Recorded on the audit row.") @click.option("--attempt", default=None, help="Caller-supplied idempotency token.") @click.option("--banner", default=None, hidden=True, help=_DEPRECATED_BANNER_HELP) @@ -1214,15 +1253,18 @@ def message_revise(ctx, msg_uuid, verdict, rationale, score) -> None: @message_group.command("revisions") @click.argument("msg_uuid") +@click.option("--limit", default=None, type=click.IntRange(1, 1000), + help="Maximum revisions to return; inspect revisions_truncated for incomplete history.") @pass_context -def message_revisions(ctx, msg_uuid) -> None: +def message_revisions(ctx, msg_uuid, limit) -> None: """The verdict revision history for a message, oldest first. \b Example: limacharlie mailsec message revisions 0057db2b-... """ - _output(ctx, _get_mailsec(ctx).list_revisions(msg_uuid)) + kwargs = {"limit": limit} if limit is not None else {} + _output(ctx, _get_mailsec(ctx).list_revisions(msg_uuid, **kwargs)) @message_group.command("bulk-action") diff --git a/limacharlie/discovery.py b/limacharlie/discovery.py index ada45a9c..acf756fe 100644 --- a/limacharlie/discovery.py +++ b/limacharlie/discovery.py @@ -208,7 +208,7 @@ ], }, "cloud_security": { - "description": "Cloud posture, identity, findings, attack paths, and vulnerability management", + "description": "Cloud and Code Security: onboarding, posture, identity, findings, code-to-cloud risk, and remediation", "commands": [ "cloudsec overview", "cloudsec changes", "cloudsec risk-trend", "cloudsec scan-status", "cloudsec topology", "cloudsec free-tier", @@ -246,7 +246,7 @@ "cloudsec resolve sensors", "cloudsec resolve assets", "cloudsec caasm assets", "cloudsec caasm coverage", "cloudsec caasm policy get", "cloudsec caasm policy set", "cloudsec caasm ingest", - "cloudsec provider manifest", "cloudsec provider test", + "cloudsec provider manifest", "cloudsec provider test", "cloudsec provider m365-certificate", "cloudsec policy vocabulary", "cloudsec policy suggest", "cloudsec simulate resources", "cloudsec simulate findings", "cloudsec export findings", "cloudsec export inventory", @@ -261,7 +261,7 @@ ], }, "email_security": { - "description": "Email security: mail triage queue, verdicts, campaigns, remediation, and reports", + "description": "Email security: provider onboarding, mailbox coverage, mail triage, verdicts, campaigns, remediation, and reports", "commands": [ "mailsec coverage", "mailsec message list", "mailsec message get", "mailsec message eml", @@ -363,6 +363,6 @@ def format_discovery(profile_name: str | None = None) -> str: lines.append(f" Commands: {len(info['commands'])}") lines.append("") - lines.append("Use 'limacharlie discover --profile ' for command details.") + lines.append("Use 'limacharlie help discover --profile ' for command details.") lines.append("Use 'limacharlie help ' for concept guides.") return "\n".join(lines) diff --git a/limacharlie/help_topics.py b/limacharlie/help_topics.py index 430e990b..028d873b 100644 --- a/limacharlie/help_topics.py +++ b/limacharlie/help_topics.py @@ -625,11 +625,240 @@ """ +# --------------------------------------------------------------------------- +# Product onboarding guides +# --------------------------------------------------------------------------- + +HELP_TOPICS["cloud-security"] = """\ +Cloud Security +============== + +Cloud Security connects your cloud, identity, and SaaS providers and turns their +inventory into findings, attack paths, access reviews, and compliance reports. +Code Security uses the same subscription and findings worklist. + +Start with one provider and a small scope: + 1. Authenticate: limacharlie auth login + 2. Find your organization UUID: limacharlie org list + 3. Select it: limacharlie auth use-org + 4. Enable: limacharlie extension subscribe --name ext-cloud-security + 5. Follow the provider-specific credential and permission guide: + https://docs.limacharlie.io/cloud-security/provider-setup/ + 6. Store credentials in the secret hive. Reference that secret from your + cloudsec_provider record as hive://secret/. + 7. Preflight your prepared provider JSON before saving: + limacharlie cloudsec provider test --input-file provider.json + 8. Save the validated connection and enable it: + limacharlie hive set --hive-name cloudsec_provider --key pilot --input-file provider.json --enabled + 9. Check collection and then review findings: + limacharlie cloudsec scan-status + limacharlie cloudsec finding list --severity CRITICAL --severity HIGH + +Access: cloudsec.get reads the product; cloudsec.set changes ordinary product +state; cloudsec.respond governs AutoFix and remediation decisions. Connection +setup also requires cloudsec_provider.* and secret.* hive permissions; policy, +saved-query, and code-rule records use cloudsec.get/set. Refresh authentication +after access changes. + +A successful save is not proof of complete collection. Read scan-status for +permission failures and coverage gaps, and provider manifest for supported +resource types. An empty worklist during the first sweep is not an all-clear. +List commands return a page; keep filters unchanged when reusing --cursor. +Commands offering --all can walk every page. cloudsec free-tier shows limits. + +Next: limacharlie help code-security +Commands: limacharlie help discover --profile cloud_security +Quick reference: limacharlie help cheatsheet --name cloud-security +""" + +HELP_TOPICS["code-security"] = """\ +Code Security +============= + +Code Security scans repositories and container images and connects code risk to +cloud resources and running workloads where evidence supports that connection. +It uses the Cloud Security subscription, permissions, and finding worklist. + +First scan: + 1. Follow https://docs.limacharlie.io/cloud-security/code-security/getting-started/ + to connect GitHub, GitLab, or Bitbucket in cloudsec_provider and configure + a cloudsec_policy record of type code_scanning. Keep the initial scope small. + 2. Check scanning, provider capabilities, and the indexed repositories: + limacharlie cloudsec code status + limacharlie cloudsec code capabilities + limacharlie cloudsec code repos + 3. Trigger a scan for a connected repository: + limacharlie cloudsec code rescan owner/repository + 4. Inspect findings, scan coverage, and image inventory: + limacharlie cloudsec finding list --repo owner/repository + limacharlie cloudsec code coverage + limacharlie cloudsec image list + +Bring your own scanner: cloudsec code ingest accepts supported scan results; +use its --help for the supported formats and required repository/commit fields. +cloudsec code provenance push accepts build evidence. cloudsec code iac-map +extract creates a sanitized map locally; never upload raw Terraform state or +plan JSON. Push the sanitized map and inspect its receipt with iac-map status. + +Read finding chain and code impact for code-to-cloud evidence. Unknown, stale, +conflicting, or unmatched evidence is a coverage gap, never proof of safety. +Some capabilities depend on your provider and deployment; inspect capabilities +before configuring PR checks or AutoFix. AutoFix and containment follow a +governed remediation run; inspect the result and its verification before closure. + +Commands: limacharlie help discover --profile cloud_security +Quick reference: limacharlie help cheatsheet --name code-security +""" + +HELP_TOPICS["email-security"] = """\ +Email Security (MailSec) +======================= + +Email Security connects Microsoft 365 or Google Workspace without changing MX +records. It judges messages, explains verdicts, groups campaigns, and supports +analyst remediation and user reports. + +Pilot onboarding: + 1. Authenticate, find your organization UUID, and select it: + limacharlie auth login + limacharlie org list + limacharlie auth use-org + 2. Enable: limacharlie extension subscribe --name ext-email-security + 3. Read the provider setup instructions: + limacharlie mailsec onboarding --provider m365 + limacharlie mailsec onboarding --provider gworkspace + https://docs.limacharlie.io/email-security/provider-setup/microsoft-365/ + https://docs.limacharlie.io/email-security/provider-setup/google-workspace/ + 4. Save the provider credential in the secret hive, then create an enabled + mailsec_provider record referencing hive://secret/. Set explicit pilot + addresses in scope.include_addresses; empty include lists cover all + discovered mailboxes. Google Workspace also needs domain-wide delegation + and Pub/Sub configuration, not just a service-account key. + 5. Test the SAVED connection by record name: + limacharlie mailsec connection test pilot + 6. Check coverage and the first messages: + limacharlie mailsec coverage + limacharlie mailsec message list + limacharlie mailsec message get + +Access: mailsec.get reads structured mail; mailsec.set changes triage state and +rules; mailsec.act remediates provider mail, revises verdicts, and tests +connections. Original-byte downloads require both mailsec.get and mailsec.get.eml +with an audited justification. Setup additionally needs +mailsec_provider.* and secret.* hive permissions; mailsec_policy and dr-mail +records use mailsec.get/set. + +Start in alert_only and review coverage before enabling automated actions. +Manual provider actions in alert_only require an explicit --force override; +the action audit trail records the outcome. Bulk and campaign actions first +preview the selection, then require the returned --confirm token to execute. +Raw EML can be binary: write it with message eml --out-file suspect.eml and +--justification "incident investigation". message get serves the judged parsed +message when available and identifies any fallback or unavailable content. + +mailsec analyze checks a local EML without ingesting it. Historical hunts use +the platform search commands; there is no separate mailsec hunt job command. +List results are paginated: reuse --cursor with the same filters to continue. + +Commands: limacharlie help discover --profile email_security +Quick reference: limacharlie help cheatsheet --name email-security +""" + # --------------------------------------------------------------------------- # Cheatsheets registry # --------------------------------------------------------------------------- CHEATSHEETS = {} +CHEATSHEETS["cloud-security"] = """\ +Cloud Security Cheatsheet +========================= + +# Set --oid on commands or select the organization once +limacharlie auth use-org +limacharlie extension subscribe --name ext-cloud-security + +# Validate and enable a prepared provider record (see the onboarding topic) +limacharlie cloudsec provider test --input-file provider.json +limacharlie hive set --hive-name cloudsec_provider --key pilot --input-file provider.json --enabled + +# Verify collection and inspect supported coverage +limacharlie cloudsec scan-status +limacharlie cloudsec provider manifest --type gcp +limacharlie cloudsec free-tier + +# Read findings and inventory +limacharlie cloudsec finding list --severity CRITICAL --severity HIGH +limacharlie cloudsec finding facets +limacharlie cloudsec inventory list --provider gcp +limacharlie cloudsec overview + +# Inspect one finding and set its analyst disposition +limacharlie cloudsec finding get +limacharlie cloudsec finding resolve --kind mitigated --reason "configuration corrected" + +Full onboarding: limacharlie help cloud-security +""" + +CHEATSHEETS["code-security"] = """\ +Code Security Cheatsheet +======================== + +# After connecting your SCM provider and enabling code_scanning policy +limacharlie cloudsec code capabilities +limacharlie cloudsec code status +limacharlie cloudsec code repos +limacharlie cloudsec code rescan owner/repository +limacharlie cloudsec finding list --repo owner/repository +limacharlie cloudsec finding chain +limacharlie cloudsec code coverage +limacharlie cloudsec image list + +# Build evidence, sanitized IaC mapping, and scanner-result ingestion +limacharlie cloudsec code provenance push --help +limacharlie cloudsec code iac-map extract --help +limacharlie cloudsec code iac-map push --help +limacharlie cloudsec code ingest --help + +# Governed fixes: inspect capabilities and run outcomes before acting +limacharlie cloudsec code autofix --help +limacharlie cloudsec remediation list + +Full onboarding: limacharlie help code-security +""" + +CHEATSHEETS["email-security"] = """\ +Email Security Cheatsheet +========================= + +# Enable the product and read the provider setup guide +limacharlie extension subscribe --name ext-email-security +limacharlie mailsec onboarding --provider m365 +limacharlie mailsec onboarding --provider gworkspace + +# After saving the credential and enabled mailsec_provider record +limacharlie mailsec connection test pilot +limacharlie mailsec coverage +limacharlie mailsec message list +limacharlie mailsec message list --verdict suspicious --verdict malicious +limacharlie mailsec message get +limacharlie mailsec message similar +limacharlie mailsec message revisions + +# Save original bytes with a justification, or analyze a local EML +limacharlie mailsec message eml --justification "incident investigation" --out-file suspect.eml +limacharlie mailsec analyze --file suspect.eml + +# Revise a verdict, then explicitly remediate (alert_only needs --force) +limacharlie mailsec message revise --verdict malicious --rationale "confirmed credential harvest" +limacharlie mailsec message action --action quarantine_message --reason "confirmed phishing" + +# Read campaign and user-report queues +limacharlie mailsec campaign list +limacharlie mailsec report list --status open + +Full onboarding: limacharlie help email-security +""" + CHEATSHEETS["common-operations"] = """\ Common Operations Cheatsheet ============================= diff --git a/limacharlie/sdk/cloudsec.py b/limacharlie/sdk/cloudsec.py index 91038c5f..158ae03f 100644 --- a/limacharlie/sdk/cloudsec.py +++ b/limacharlie/sdk/cloudsec.py @@ -18,7 +18,7 @@ The AppSec code lane additionally exposes what each connected source-control organization may actually DO -(:meth:`CloudSec.get_code_capabilities` — GitHub connections only), the +(:meth:`CloudSec.get_code_capabilities`), the dependency-upgrade queue (:meth:`CloudSec.get_code_fixes`), the registry-backed container-image inventory (:meth:`CloudSec.list_image_repos`, :meth:`CloudSec.list_container_images`, @@ -80,6 +80,29 @@ from .organization import Organization +_LINEAGE_STATUSES = frozenset(("verified", "asserted", "inferred", "ambiguous", "unknown")) + + +def _lineage_values(values: list[str] | None) -> list[str] | None: + if values is None: + return None + if not isinstance(values, (list, tuple)) or not values: + raise ValueError("lineage_status must be a non-empty list of effective lineage states") + if any(not isinstance(value, str) or value.strip() not in _LINEAGE_STATUSES for value in values): + raise ValueError("invalid lineage_status: use verified, asserted, inferred, ambiguous or unknown") + return sorted({value.strip() for value in values}) + + +def _validate_csv_chunk(max_rows: int | None, cursor: str | None) -> None: + if max_rows is not None and ( + isinstance(max_rows, bool) or not isinstance(max_rows, int) + or not 1 <= max_rows <= 100000 + ): + raise ValueError("max_rows must be an integer between 1 and 100000") + if cursor and max_rows is None: + raise ValueError("resuming a CSV export with cursor requires max_rows") + + def _add_pairs( pairs: list[tuple[str, str]], key: str, @@ -2483,9 +2506,9 @@ def check_pull_request( self, repo: str, pr: int, - base_sha: str, - head_sha: str, - action: str, + base_sha: str | None = None, + head_sha: str | None = None, + action: str | None = None, *, prev_base_sha: str | None = None, base_ref: str | None = None, @@ -2495,7 +2518,7 @@ def check_pull_request( """Ask the code lane what a pull request INTRODUCES, as a check run. The lane scans the pull request's base and head and publishes a - GitHub check run on the head commit reporting only what is NEW in + provider check on the head commit reporting only what is NEW in it; the repository's own findings stay on :meth:`list_code_repos`. Its normal caller is the shipped D&R rule on the org's source-control webhook, but this is a plain @@ -2505,16 +2528,18 @@ def check_pull_request( repo: The repository — ``"/"``, its bare name, or its canonical urn. pr: The pull-request number. - base_sha: A FULL commit id. Still required, but NOT - authoritative: the lane takes the base from the provider, + base_sha: A FULL commit id. Required for GitHub, optional for + GitLab.com and Bitbucket Cloud. Never authoritative: the lane + takes the base from the provider, because a caller-chosen base decides what the diff is measured from and a base equal to the head would make any pull request look like it introduced nothing. - head_sha: The FULL commit id of the head. A branch or tag name + head_sha: The FULL commit id of the head (Bitbucket also accepts + 12-39 hexadecimal characters). A branch or tag name is refused — the check is published ON the commit, and a ref would let it be attached to a commit nobody proposed. action: REQUIRED. The webhook action: ``opened``, - ``synchronize``, ``reopened`` or ``edited``. Every other + ``synchronize``, ``reopened`` or ``edited`` (GitHub only). Every other pull-request event leaves what the pull request introduces untouched and is refused — and so is an ABSENT action: the collection host checks membership of that closed set @@ -2537,6 +2562,10 @@ def check_pull_request( head_ref: Optional branch the pull request comes from. provider: Source-control provider; defaults to ``github``. + Raises: + TypeError: If head_sha or action is missing. + ValueError: If GitHub lacks base_sha or another provider uses edited. + Returns: ``{"accepted": bool, "repo": str, "pr": int, "provider": str, "debounce_seconds": int}``. @@ -2566,12 +2595,20 @@ def check_pull_request( The verdict is the check run's conclusion, set by the policy's ``gating.fail_on``. """ + selected_provider = (provider or "github").strip().lower() + if not head_sha or not action: + raise TypeError("head_sha and action are required") + if selected_provider not in ("gitlab", "bitbucket") and not base_sha: + raise ValueError("base_sha is required for GitHub pull-request checks") + if selected_provider in ("gitlab", "bitbucket") and action == "edited": + raise ValueError("edited is a GitHub-only action") body: dict[str, Any] = { "repo": repo, "pr": pr, - "base_sha": base_sha, "head_sha": head_sha, } + if base_sha is not None: + body["base_sha"] = base_sha body["action"] = action for key, value in ( ("prev_base_sha", prev_base_sha), @@ -2700,19 +2737,16 @@ def autofix_code_finding( ``pr_merged_unverifiable``; a PR closed without merge ends with ``pr_closed``. Neither is a verified fix. - For **npm** the ``package-lock.json`` *is* rewritten by - default: one read-only registry metadata document supplies the - new version's resolved URL and integrity digest. It is left - stale only where the ``code_scanning`` policy sets - ``autofix_registry_access: false``, where the lock is a - ``yarn.lock``/``pnpm-lock.yaml``, or where the entry could not - be rewritten safely. For **go** the ``go.sum`` is *not* - regenerated — it hashes a module zip nobody downloaded — and - only where the tree has one. For **pip** - (``requirements.txt``) and **maven** there is no lockfile, so - the change is complete. Wherever a lock is left stale the pull - request says so prominently and names the command to run; trust - the pull request over this summary. + For **npm**, lockfiles beside the manifest (package-lock.json, + npm-shrinkwrap.json, yarn.lock or pnpm-lock.yaml) are updated + when supported. A yarn or pnpm lock that cannot be rewritten + safely is refused before a job runs. A package-lock that cannot + be updated, for example when ``autofix_registry_access`` is false, + is flagged ``lockfile_stale`` with the command to run in the PR. + For **go**, ``go.sum`` is updated from verified checksum-database + entries; a missing or uncompletable go.sum refuses the fix instead + of opening a PR that does not build. Trust the PR's recorded + outcome and limitations when reviewing the proposed fix. """ body: dict[str, Any] = {"finding_id": finding_id} if repo: @@ -3117,12 +3151,10 @@ def get_code_capabilities(self, *, repo: str | None = None) -> dict[str, Any]: for the AppSec code lane: repository scanning, PR checks, PR comments, and dependency AutoFix pull requests. - This only covers **GitHub** connections. A GitLab or Bitbucket - connection scans with its own read-only token and holds no write - plane to detect (no PR checks, no PR comments, no AutoFix pull - requests), so it never appears here — not even as an ``unknown`` - entry. Use ``cloudsec provider manifest`` for what a GitLab or - Bitbucket connection actually collects. + GitHub connections are always reported. GitLab.com and Bitbucket + Cloud connections appear when their workflow support is enabled in + this deployment; absence does not mean repository scanning is off. + Use ``cloudsec provider manifest`` to inspect collection coverage. A capability reading ``available`` means the control MAY be offered, never that anything fires on its own: a ``pr_checks`` @@ -3132,7 +3164,7 @@ def get_code_capabilities(self, *, repo: str | None = None) -> dict[str, Any]: Args: repo: Narrow to the one connection covering a single repository (``"/"`` as :meth:`list_code_repos` returns - it). Omit to list every GitHub connection. + it). Omit to list enabled workflow connections. Returns: ``{"connections": [{"connection", "org", "provider", "mode", @@ -3321,6 +3353,7 @@ def get_image_repo_facets( has_findings: bool | None = None, has_images: bool | None = None, scanning_state: str | None = None, + lineage_facet: bool | None = None, ) -> dict[str, Any]: """Cross-filtered facet counts for the image-repository list. @@ -3328,10 +3361,18 @@ def get_image_repo_facets( paging, which this endpoint ignores), so the rail describes the population that list returns. + Args: + q, provider, account, registry, region, has_findings, has_images, + scanning_state: See :meth:`list_image_repos`. + lineage_facet: Include digest-global effective lineage counts. + Repository selectors do not constrain these counts. + Returns: ``{"total": int, "providers": [{"value", "count"}, ...], "accounts": [...], "registries": [...], - "scanning_states": [...]}``. + "scanning_states": [...]}``. With ``lineage_facet=True``, also + includes ``lineage_statuses`` matching the image lineage filters; + stale decisions count as ``unknown``. Note: Each faceted dimension excludes its OWN selector so the rail @@ -3349,6 +3390,7 @@ def get_image_repo_facets( q=q, provider=provider, account=account, registry=registry, region=region, has_findings=has_findings, has_images=has_images, scanning_state=scanning_state, + lineage_facet=lineage_facet, )) def list_container_images( @@ -3363,6 +3405,7 @@ def list_container_images( findings: str | None = None, running: bool | None = None, signed: bool | None = None, + lineage_status: list[str] | None = None, sort: str | None = None, order: str | None = None, cursor: str | None = None, @@ -3396,6 +3439,10 @@ def list_container_images( provider reports it: an image whose status is UNKNOWN matches NEITHER ``True`` nor ``False``, and the field is not echoed back in the row. + lineage_status: Effective source-lineage states, OR'd: + ``verified``, ``asserted``, ``inferred``, ``ambiguous`` or + ``unknown``. Stale decisions match ``unknown``. This is + separate from image-signature status. sort: ``name`` (the default), ``risk`` or ``pushed``. An unrecognised key is silently coerced to ``name``. order: ``asc`` or ``desc`` (default ``asc`` for ``name``, @@ -3403,12 +3450,16 @@ def list_container_images( cursor: Keyset-pagination token from a previous page. limit: Page size (default 100, max 1000). + Raises: + ValueError: If lineage_status contains an invalid state. + RuntimeError: If the server does not acknowledge the lineage filter. + Returns: ``{"images": [{"urn", "digest", "name", "open_findings", "findings_by_severity", "top_severity", "workload_count", "built_from_repo_count", "repository_count", "repositories", "repositories_truncated", "first_seen", "last_seen", - "scanner_provenance", "registry_observed"}, ...], + "scanner_provenance", "registry_observed", "lineage"}, ...], "next_cursor": str, "total": int, "coverage": {...}}``. ``urn`` is what :meth:`list_findings` takes as ``image_urn``. @@ -3437,12 +3488,20 @@ def list_container_images( rollup rebuilt once per collection pass, so they describe the last rebuild rather than this instant. """ - return self._get("code/images", _query_pairs( + lineage_status = _lineage_values(lineage_status) + response = self._get("code/images", _query_pairs( q=q, repo_urn=repo_urn, provider=provider, account=account, registry=registry, tag=tag, findings=findings, - running=running, signed=signed, sort=sort, order=order, + running=running, signed=signed, lineage_status=lineage_status, + sort=sort, order=order, cursor=cursor, limit=limit, )) + if lineage_status is not None and ( + not isinstance(response, dict) + or response.get("applied_lineage_status") != lineage_status + ): + raise RuntimeError("lineage_status was not acknowledged by this server version") + return response def iter_container_images(self, **selectors: Any): """Yield every matching container image, page by page. @@ -3618,16 +3677,39 @@ def list_caasm_assets( self, *, q: str | None = None, + kind: list[str] | None = None, + source: list[str] | None = None, + posture_encryption: list[str] | None = None, + posture_screen_lock: list[str] | None = None, + posture_compromised: list[str] | None = None, + posture_managed: list[str] | None = None, + sort: str | None = None, cursor: str | None = None, limit: int | None = None, ) -> dict[str, Any]: """The merged third-party asset inventory (EDR/IdP/MDM/scanner sources). + Args: + q: Substring search over asset identifiers. + kind: Asset kinds, OR'd (for example ``device`` or ``user``). + source: Observing tool identifiers, OR'd. + posture_encryption, posture_screen_lock, posture_compromised, + posture_managed: Reported posture values, OR'd. An empty + string selects assets where no source reported the fact. + sort: ``urn`` (stable default) or ``last_seen`` (newest first). + cursor: Opaque cursor from the previous page. + limit: Page size. Follow ``next_cursor`` until absent. + Returns: ``{"resources": [...], "next_cursor": str}``. """ return self._get("caasm/assets", _query_pairs( - q=q, cursor=cursor, limit=limit, + q=q, kind=kind, source=source, + posture_encryption=posture_encryption, + posture_screen_lock=posture_screen_lock, + posture_compromised=posture_compromised, + posture_managed=posture_managed, sort=sort, + cursor=cursor, limit=limit, )) def list_caasm_coverage( @@ -3761,6 +3843,35 @@ def get_free_tier(self) -> dict[str, Any]: """ return self._get("free-tier") + def mint_m365_certificate( + self, connection: str, *, client_id: str | None = None, + replace: bool = False, + ) -> dict[str, Any]: + """Generate a connection's Microsoft 365 certificate credential. + + Requires both ``cloudsec.set`` and ``secret.set``. The private key is + stored in the organization's secret store and never returned. Repeat + calls return the existing public certificate unless replacing it. + + Args: + connection: The ``cloudsec_provider`` record name, also used to + name the managed secret. The connection may be saved later. + client_id: Optional Entra application's client id (GUID). + replace: Replace the stored key pair immediately. An existing + connection may stop authenticating until you upload the new + public certificate to the Entra app registration. + + Returns: + dict: ``created``, ``credentials`` (a Hive secret reference), + ``secret_name``, ``certificate`` (base64 DER), ``certificate_pem``, + ``thumbprint`` and certificate validity timestamps. Upload the + public certificate to the Entra application's certificates. + """ + body: dict[str, Any] = {"connection": connection, "replace": replace} + if client_id is not None: + body["client_id"] = client_id + return self._post("providers/m365/certificate", body) + def test_provider(self, provider: dict[str, Any]) -> dict[str, Any]: """Preflight a cloud provider configuration before saving it. @@ -3903,6 +4014,8 @@ def export_findings_csv( q: str | None = None, sort: str | None = None, order: str | None = None, + max_rows: int | None = None, + cursor: str | None = None, ) -> str: """Export the (filtered) findings worklist as CSV text. @@ -3913,6 +4026,14 @@ def export_findings_csv( server walks the full filtered set (no pagination), capped at 100k rows. + Args: + max_rows: Bound this export to roughly this many rows (1-100000, + rounded up to full 1000-row pages). A remaining page is + indicated by a trailing ``# next_cursor=`` row. + cursor: Resume token from a bounded export; requires max_rows. + Keep all filter and sort selectors unchanged when resuming. + Other selectors: See the corresponding list method. + Returns: The CSV document as a string. """ @@ -3925,6 +4046,8 @@ def export_findings_csv( source=source, reachable=reachable, kev=kev, q=q, sort=sort, order=order, ) + _validate_csv_chunk(max_rows, cursor) + pairs.extend(_query_pairs(max_rows=max_rows, cursor=cursor)) pairs.append(("format", "csv")) return self._get("findings", pairs, raw_response=True) @@ -3939,6 +4062,8 @@ def export_inventory_csv( q: str | None = None, account_empty: bool | None = None, account_unscoped: bool | None = None, + max_rows: int | None = None, + cursor: str | None = None, ) -> str: """Export the (filtered) cloud resource inventory as CSV text. @@ -3946,6 +4071,14 @@ def export_inventory_csv( server walks the full filtered set (no pagination), capped at 100k rows. + Args: + max_rows: Bound this export to roughly this many rows (1-100000, + rounded up to full 1000-row pages). A remaining page is + indicated by a trailing ``# next_cursor=`` row. + cursor: Resume token from a bounded export; requires max_rows. + Keep all filter and sort selectors unchanged when resuming. + Other selectors: See the corresponding list method. + Returns: The CSV document as a string. """ @@ -3956,6 +4089,8 @@ def export_inventory_csv( region=region, q=q, has_iac_origin=has_iac_origin, **selector, ) + _validate_csv_chunk(max_rows, cursor) + pairs.extend(_query_pairs(max_rows=max_rows, cursor=cursor)) pairs.append(("format", "csv")) return self._get("inventory", pairs, raw_response=True) diff --git a/limacharlie/sdk/mailsec.py b/limacharlie/sdk/mailsec.py index f7630eb6..fb4d19ab 100644 --- a/limacharlie/sdk/mailsec.py +++ b/limacharlie/sdk/mailsec.py @@ -294,22 +294,34 @@ def _delete( # Coverage # ------------------------------------------------------------------ - def get_coverage(self, *, window_days: int | None = None) -> dict[str, Any]: + def get_coverage( + self, *, window_days: int | None = None, + since: str | None = None, until: str | None = None, + ) -> dict[str, Any]: """Coverage and volume for the org: mailboxes protected vs not, and what was analysed over the window. Args: window_days: Days of volume to summarise (server default applies - when omitted). + when omitted). Cannot be combined with since or until. + since: Start of the volume window (RFC3339 or unix seconds). + until: End of the volume window (RFC3339 or unix seconds). Returns: The coverage summary, including the mailbox states that are NOT protected — a mailbox we cannot subscribe is reported as broken rather than omitted, so the number is a coverage statement rather than a count of what happened to work. + + Raises: + ValueError: If window_days is combined with since or until. """ + if window_days is not None and (since is not None or until is not None): + raise ValueError("window_days cannot be combined with since or until") pairs: list[tuple[str, str]] = [] _add_scalar(pairs, "window_days", window_days) + _add_scalar(pairs, "since", since) + _add_scalar(pairs, "until", until) return self._get("coverage", pairs) # ------------------------------------------------------------------ @@ -326,6 +338,7 @@ def list_messages( campaign_id: str | None = None, state: list[str] | None = None, direction: list[str] | None = None, + lane: str | None = None, user_reported: bool | None = None, min_score: int | None = None, link_domain: str | None = None, @@ -351,6 +364,10 @@ def list_messages( campaign_id: Only members of one campaign. state: Message lifecycle state (repeatable). direction: ``inbound``, ``outbound``, ``internal`` (repeatable). + lane: ``live`` or ``backfill``. Omit to include either processing + lane. Supported with time, verdict and IOC queries; combining + it with mailbox, sender_email or campaign_id is refused by + the server with ``lane_unsupported``. user_reported: Tri-state. ``True`` only reported mail, ``False`` only unreported, ``None`` (default) either. A human reporting a message is the strongest signal the product gets, so this @@ -404,22 +421,27 @@ def list_messages( ("cursor", cursor), ("limit", limit), ("user_reported", user_reported), + ("lane", lane), ): _add_scalar(pairs, key, val) return self._get("messages", pairs) def get_message(self, msg_uuid: str) -> dict[str, Any]: - """One message: the index row plus the re-parsed MDM (the drawer). + """Get the message index row and parsed Message Data Model (MDM). - The MDM is not stored in Spanner — the index row is a summary and the - raw bytes live in object storage — so the drawer re-parses the stored - EML and says which path produced it (``mdm_source``). Enrichments are - deliberately ABSENT rather than recomputed: they were resolved against - sender profiles as they existed at ingest, and synthesising today's - values would show a reputation the verdict was never based on. + ``mdm_source: stored`` serves the preserved MDM used to judge the + message, including its original enrichments. ``eml_reparse`` is a + fallback that parses the retained EML with today's parser and leaves + enrichments absent. Expired content yields ``mdm: null`` and an + ``mdm_unavailable_reason`` while the index row remains available. - An unknown id returns ``{"message": None}`` rather than an error: the - index has a 35-day TTL, so a miss is a normal outcome. + Args: + msg_uuid: Message UUID. + + Returns: + Message detail, or ``{"message": None}`` for an unknown or + expired UUID. Message-index retention is at most 35 days and may + be shortened by the organization's retention policy. """ return self._get(f"messages/{_seg(msg_uuid)}") @@ -465,12 +487,28 @@ def list_similar_messages( cursor: str | None = None, limit: int | None = None, ) -> dict[str, Any]: - """Messages clustered with this one — the campaign view from a single - message, which is how "who else got this" is answered.""" - pairs: list[tuple[str, str]] = [] - _add_scalar(pairs, "cursor", cursor) - _add_scalar(pairs, "limit", limit) - return self._get(f"messages/{_seg(msg_uuid)}/similar", pairs) + """Get recent messages sharing clustering keys with this message. + + These are candidates, not necessarily members of the same campaign. + The response names matched keys and its lookback windows. The route + is not paginated; the old cursor and limit arguments were ignored + by the server and are now refused when supplied. + + Args: + msg_uuid: Message UUID. + cursor: Unsupported legacy parameter; leave unset. + limit: Unsupported legacy parameter; leave unset. + + Raises: + ValueError: If cursor or limit is supplied. Use list_messages + with campaign_id or time/IOC filters for a paginated search. + """ + if cursor is not None or limit is not None: + raise ValueError( + "similar messages are not paginated; omit cursor and limit, " + "or use list_messages with campaign_id or time/IOC filters" + ) + return self._get(f"messages/{_seg(msg_uuid)}/similar") def act_on_message( self, @@ -590,15 +628,25 @@ def revise_verdict( body["score"] = score return self._post(f"messages/{_seg(msg_uuid)}/verdict", body) - def list_revisions(self, msg_uuid: str) -> dict[str, Any]: + def list_revisions(self, msg_uuid: str, *, limit: int | None = None) -> dict[str, Any]: """The verdict revision history for one message, oldest first. Requires ``mailsec.get``. Every entry carries its ``seq``, the ``mode`` and ``actor`` that decided it, the ``verdict`` it set, its ``decided_at`` time, and the ``rationale`` given — the audit of how a message's disposition moved over time, read from the bottom up. + + Args: + msg_uuid: Message UUID. + limit: Maximum revisions to return (1–1000 at the gateway). + + Returns: + Revision entries and ``revisions_truncated``. There is no cursor; + a truncated response is an incomplete history. """ - return self._get(f"messages/{_seg(msg_uuid)}/revisions") + pairs: list[tuple[str, str]] = [] + _add_scalar(pairs, "limit", limit) + return self._get(f"messages/{_seg(msg_uuid)}/revisions", pairs) # ------------------------------------------------------------------ # Bulk remediation by message id @@ -1178,17 +1226,33 @@ def test_connection(self, record: str, *, include_watch: bool = False) -> dict[s body["include_watch"] = True return self._post(f"connections/{_seg(record)}/test", body) - def get_onboarding(self, *, provider: str | None = None) -> dict[str, Any]: - """The setup guide for connecting a mail provider, with this org's own - values already substituted in. + def get_onboarding( + self, *, provider: str | None = None, project_id: str | None = None, + sa_email: str | None = None, topic: str | None = None, + subscription: str | None = None, + ) -> dict[str, Any]: + """Get the provider setup guide with optional Workspace substitutions. + + The backend supplies current scopes and setup steps. Workspace + commands contain placeholders until the customer supplies their own + project and service account; this read creates no provider resources. - Served by the backend rather than written into the docs or the web app - so that the identifiers a customer must paste — the service account, - the topic, the subscription — are the real ones for this deployment - rather than placeholders a reader has to translate. + Args: + provider: ``gworkspace`` (default) or ``m365``. + project_id: Customer's Google Cloud project ID. + sa_email: Customer's service account email address. + topic: Workspace Pub/Sub topic name override. + subscription: Workspace Pub/Sub subscription name override. + + Returns: + Current provider scopes, setup steps and Workspace setup script. """ pairs: list[tuple[str, str]] = [] _add_scalar(pairs, "provider", provider) + _add_scalar(pairs, "project_id", project_id) + _add_scalar(pairs, "sa_email", sa_email) + _add_scalar(pairs, "topic", topic) + _add_scalar(pairs, "subscription", subscription) return self._get("onboarding", pairs) # ------------------------------------------------------------------ diff --git a/tests/unit/test_cli_cloudsec.py b/tests/unit/test_cli_cloudsec.py index d889433e..0c627ea3 100644 --- a/tests/unit/test_cli_cloudsec.py +++ b/tests/unit/test_cli_cloudsec.py @@ -48,7 +48,7 @@ def _invoke(args, mock_cs_cls, return_value=None, stdin=None): "resolve_sensors", "resolve_assets", "list_caasm_assets", "list_caasm_coverage", "get_caasm_policy", "set_caasm_policy", "caasm_ingest", - "test_provider", "get_provider_manifests", "get_fleet_overview", + "test_provider", "mint_m365_certificate", "get_provider_manifests", "get_fleet_overview", "get_policy_vocabulary", "suggest_policy_values", "simulate_resource_match", "simulate_finding_match", "list_code_repos", "get_code_status", "get_code_sbom", @@ -958,6 +958,7 @@ def test_export_findings_stdout(self): exploit_band=None, grain=None, cause=None, source=None, owner=None, sla=None, reachable=None, kev=None, q=None, sort=None, order=None, + max_rows=None, cursor=None, ) def test_export_findings_to_file(self, tmp_path): @@ -982,6 +983,7 @@ def test_export_inventory(self): has_iac_origin=None, resource_type="Bucket", provider="gcp", account=None, region=None, q=None, account_empty=None, + max_rows=None, cursor=None, ) def test_export_compliance(self): @@ -1110,6 +1112,7 @@ def test_export_all_accounts_flag(self): has_iac_origin=None, resource_type=None, provider=None, account=None, region=None, q=None, account_empty=None, + max_rows=None, cursor=None, ) def test_export_account_empty_flag(self): @@ -1123,6 +1126,7 @@ def test_export_account_empty_flag(self): has_iac_origin=None, resource_type=None, provider=None, account=None, region=None, q=None, account_empty=True, + max_rows=None, cursor=None, ) def test_account_scope_flags_are_mutually_exclusive(self): @@ -2027,7 +2031,7 @@ def test_code_scan_sast_runs_the_default_rules_by_default(self, tmp_path, monkey cmd = cap["cmd"] assert cmd[0] == "docker" assert cs_mod.DEFAULT_CODE_SCANNER_IMAGE in cmd - assert cs_mod.DEFAULT_CODE_SCANNER_IMAGE.endswith(":v0.16.0") + assert cs_mod.DEFAULT_CODE_SCANNER_IMAGE.endswith(":v0.24.0") assert "--default-rules" in cmd and "--rules-file" not in cmd # The pinned image is known to take the flags, so no hint is attached. assert cap["usage_hint"] is None @@ -2982,3 +2986,123 @@ def test_push_bounds_before_client(self, tmp_path): result, instance = _invoke(["cloudsec", "code", "provenance", "push", "-f", str(document)], mock) assert result.exit_code != 0 instance.push_code_provenance.assert_not_called() + + +class TestProductContractUpdates: + def test_m365_certificate_writes_only_public_der(self, tmp_path): + import base64 + path = tmp_path / "connection.cer" + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, inst = _invoke(["cloudsec", "provider", "m365-certificate", "my-entra", + "--client-id", "application-guid", "--replace", "--out", str(path)], + cls, {"certificate": base64.b64encode(b"public DER").decode(), + "credentials": "hive://secret/cloudsec-m365-my-entra"}) + assert result.exit_code == 0, result.output + inst.mint_m365_certificate.assert_called_once_with("my-entra", client_id="application-guid", replace=True) + assert path.read_bytes() == b"public DER" + assert json.loads(result.output)["credentials"].startswith("hive://secret/") + + def test_m365_certificate_default_does_not_replace(self): + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, inst = _invoke(["cloudsec", "provider", "m365-certificate", "my-entra"], cls) + assert result.exit_code == 0, result.output + inst.mint_m365_certificate.assert_called_once_with("my-entra", client_id=None, replace=False) + + def test_m365_invalid_certificate_does_not_clobber_output(self, tmp_path): + path = tmp_path / "connection.cer" + path.write_bytes(b"existing") + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, _ = _invoke(["cloudsec", "provider", "m365-certificate", "my-entra", "--out", str(path)], + cls, {"certificate": "not-base64"}) + assert result.exit_code != 0 + assert path.read_bytes() == b"existing" + + @pytest.mark.parametrize("walk_all", [False, True]) + def test_image_lineage_filters_reach_every_page(self, walk_all): + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + args = ["cloudsec", "image", "list", "--lineage-status", "unknown", "--lineage-status", "asserted"] + result, inst = _invoke(args + (["--all"] if walk_all else []), cls) + assert result.exit_code == 0, result.output + method = inst.iter_container_images if walk_all else inst.list_container_images + assert method.call_args.kwargs["lineage_status"] == ["unknown", "asserted"] + + def test_bad_lineage_status_fails_before_authentication(self): + p1, p2, p3 = _patches() + with p1 as client, p2, p3: + result = CliRunner().invoke(cli, ["cloudsec", "image", "list", "--lineage-status", "verfied"]) + assert result.exit_code != 0 + client.assert_not_called() + + @pytest.mark.parametrize("flag,value", [("--lineage-facet", True), ("--no-lineage-facet", False)]) + def test_image_lineage_facets(self, flag, value): + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, inst = _invoke(["cloudsec", "image", "repo-facets", flag], cls) + assert result.exit_code == 0, result.output + assert inst.get_image_repo_facets.call_args.kwargs["lineage_facet"] is value + + def test_caasm_empty_posture_survives_cli_parsing(self): + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, inst = _invoke(["cloudsec", "caasm", "assets", "--kind", "device", "--source", "okta", + "--posture-encryption", "", "--posture-managed", "managed", + "--posture-screen-lock", "enabled", "--posture-compromised", "false", + "--sort", "last_seen", "--cursor", "c2"], cls) + assert result.exit_code == 0, result.output + assert inst.list_caasm_assets.call_args.kwargs["posture_encryption"] == [""] + assert inst.list_caasm_assets.call_args.kwargs["sort"] == "last_seen" + assert inst.list_caasm_assets.call_args.kwargs["cursor"] == "c2" + assert inst.list_caasm_assets.call_args.kwargs["source"] == ["okta"] + + @pytest.mark.parametrize("resource", ["findings", "inventory"]) + def test_csv_resume_cursor_requires_chunk_size_before_authentication(self, resource): + p1, p2, p3 = _patches() + with p1 as client, p2, p3: + result = CliRunner().invoke(cli, ["cloudsec", "export", resource, "--cursor", "c2"]) + assert result.exit_code != 0 + assert "--max-rows" in result.output + client.assert_not_called() + + @pytest.mark.parametrize("resource", ["findings", "inventory"]) + def test_csv_chunk_options_reach_sdk(self, resource): + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, inst = _invoke(["cloudsec", "export", resource, "--max-rows", "2000", "--cursor", "c2"], cls) + assert result.exit_code == 0, result.output + assert result.output == "col_a,col_b\n1,2\n" + method = getattr(inst, "export_" + resource + "_csv") + assert method.call_args.kwargs["max_rows"] == 2000 + assert method.call_args.kwargs["cursor"] == "c2" + + +@pytest.mark.parametrize("exit_code", [125, 126, 127]) +def test_docker_start_error_gives_actionable_scanner_access_hint(exit_code): + from limacharlie.commands.cloudsec import _run + import click + with patch("limacharlie.commands.cloudsec.subprocess.run", return_value=MagicMock(returncode=exit_code)): + with pytest.raises(click.ClickException, match="registry access.*--image.*--binary"): + _run(["docker", "run", "scanner"], 60, container="test-scanner") + + +@pytest.mark.parametrize("provider", ["gitlab", "bitbucket"]) +def test_scm_pr_check_can_omit_base_sha(provider): + p1, p2, p3 = _patches() + with p1, p2, p3 as cls: + result, inst = _invoke(["cloudsec", "code", "pr-check", "acme/api", "--pr", "42", "--provider", provider, + "--head-sha", "b" * 40, "--action", "synchronize"], cls) + assert result.exit_code == 0, result.output + assert inst.check_pull_request.call_args.args == ("acme/api", 42, None, "b" * 40, "synchronize") + + +def test_github_pr_check_requires_base_before_authentication(): + p1, p2, p3 = _patches() + with p1 as client, p2, p3: + result = CliRunner().invoke(cli, ["cloudsec", "code", "pr-check", "acme/api", "--pr", "42", + "--head-sha", "b" * 40, "--action", "opened"]) + assert result.exit_code != 0 + assert "--base-sha" in result.output + client.assert_not_called() diff --git a/tests/unit/test_cli_mailsec_contract.py b/tests/unit/test_cli_mailsec_contract.py new file mode 100644 index 00000000..e8e99b93 --- /dev/null +++ b/tests/unit/test_cli_mailsec_contract.py @@ -0,0 +1,117 @@ +"""MailSec CLI arguments must reach the published REST contract unchanged.""" + +import json +from unittest.mock import MagicMock, patch + +import pytest +from click.testing import CliRunner + +from limacharlie.cli import cli +from limacharlie.sdk.mailsec import Mailsec + + +OID = "11111111-2222-3333-4444-555555555555" + + +def invoke(*args): + org = MagicMock() + org.oid = OID + org.client.request.return_value = {"messages": [], "next_cursor": ""} + # Keep the real Mailsec SDK so this covers Click parsing and serialization, + # including omission of unset values rather than just a mock method call. + with patch("limacharlie.commands.mailsec.Client"), patch( + "limacharlie.commands.mailsec.Organization", return_value=org + ): + result = CliRunner().invoke(cli, ["--oid", OID, "--output", "json", "mailsec", *args]) + return result, org.client.request + + +def query(request, path): + assert request.call_args.args == ("GET", f"mailsec/{OID}/{path}") + return request.call_args.kwargs["query_params"] + + +def test_workspace_onboarding_substitutes_customer_resource_names(): + result, request = invoke( + "onboarding", "--provider", "gworkspace", "--project-id", "customer-project", + "--sa-email", "mail@customer-project.iam.gserviceaccount.com", + "--topic", "mail-topic", "--subscription", "mail-subscription", + ) + assert result.exit_code == 0, result.output + assert dict(query(request, "onboarding")) == { + "provider": "gworkspace", "project_id": "customer-project", + "sa_email": "mail@customer-project.iam.gserviceaccount.com", + "topic": "mail-topic", "subscription": "mail-subscription", + } + assert json.loads(result.output)["messages"] == [] + + +@pytest.mark.parametrize("lane", ["live", "backfill"]) +def test_lane_is_forwarded_with_opaque_cursor_and_existing_filters(lane): + result, request = invoke( + "message", "list", "--lane", lane, "--verdict", "malicious", "--verdict", "suspicious", + "--no-user-reported", "--since", "2026-09-01T00:00:00Z", "--cursor", "opaque+/=", + ) + assert result.exit_code == 0, result.output + pairs = query(request, "messages") + assert ("lane", lane) in pairs + assert ("cursor", "opaque+/=") in pairs + assert ("user_reported", "false") in pairs + assert [value for key, value in pairs if key == "verdict"] == ["malicious", "suspicious"] + + +def test_coverage_accepts_explicit_range(): + result, request = invoke("coverage", "--since", "1788220800", "--until", "1788307200") + assert result.exit_code == 0, result.output + assert dict(query(request, "coverage")) == {"since": "1788220800", "until": "1788307200"} + + +@pytest.mark.parametrize("bound", ["--since", "--until"]) +def test_coverage_rejects_ambiguous_window_before_request(bound): + result, request = invoke("coverage", "--window-days", "7", bound, "1788220800") + assert result.exit_code == 2 + assert "cannot be combined" in result.output + request.assert_not_called() + + +def test_revision_history_limit_reaches_gateway(): + result, request = invoke("message", "revisions", "message-uuid", "--limit", "1000") + assert result.exit_code == 0, result.output + assert query(request, "messages/message-uuid/revisions") == [("limit", "1000")] + + +@pytest.mark.parametrize("args,path", [ + (("coverage",), "coverage"), + (("onboarding",), "onboarding"), + (("message", "list"), "messages"), + (("message", "revisions", "message-uuid"), "messages/message-uuid/revisions"), + (("message", "similar", "message-uuid"), "messages/message-uuid/similar"), +]) +def test_unset_options_do_not_change_default_reads(args, path): + result, request = invoke(*args) + assert result.exit_code == 0, result.output + assert query(request, path) is None + + +@pytest.mark.parametrize("option,value", [("--cursor", "previous-page"), ("--limit", "1")]) +def test_similar_refuses_unsupported_paging_instead_of_repeating_first_page(option, value): + result, request = invoke("message", "similar", "message-uuid", option, value) + assert result.exit_code == 2 + assert "not paginated" in result.output + request.assert_not_called() + + +@pytest.mark.parametrize("kwargs", [{"cursor": ""}, {"cursor": "opaque"}, {"limit": 1}]) +def test_sdk_similar_refuses_unsupported_paging(kwargs): + org = MagicMock() + with pytest.raises(ValueError, match="not paginated"): + Mailsec(org).list_similar_messages("message-uuid", **kwargs) + org.client.request.assert_not_called() + + +@pytest.mark.parametrize("bound", ["since", "until"]) +def test_sdk_coverage_rejects_ambiguous_window(bound): + org = MagicMock() + with pytest.raises(ValueError, match="cannot be combined"): + Mailsec(org).get_coverage(window_days=7, **{bound: "1788220800"}) + org.client.request.assert_not_called() diff --git a/tests/unit/test_sdk_cloudsec.py b/tests/unit/test_sdk_cloudsec.py index df12cd86..cd79e645 100644 --- a/tests/unit/test_sdk_cloudsec.py +++ b/tests/unit/test_sdk_cloudsec.py @@ -1503,3 +1503,84 @@ def test_wrong_type_rejected_without_allocation_or_network(self, cs, mock_org, d with pytest.raises(TypeError): cs.push_code_provenance(document) mock_org.client.request.assert_not_called() + + +class TestProductContractUpdates: + def test_m365_certificate_stays_org_scoped_and_preserves_replace(self, cs, mock_org): + cs.mint_m365_certificate("my-entra", client_id="application-guid", replace=True) + path, body = _post_call(mock_org) + assert path == f"cloudsec/{OID}/providers/m365/certificate" + assert body == {"connection": "my-entra", "client_id": "application-guid", "replace": True} + + def test_m365_certificate_is_idempotent_by_default(self, cs, mock_org): + cs.mint_m365_certificate("my-entra") + _, body = _post_call(mock_org) + assert body == {"connection": "my-entra", "replace": False} + + def test_lineage_filter_requires_server_receipt_and_normalizes_or_values(self, cs, mock_org): + mock_org.client.request.return_value = {"images": [], "applied_lineage_status": ["asserted", "unknown"]} + cs.list_container_images(lineage_status=["unknown", " asserted ", "unknown"], cursor="page2") + _, pairs = _get_call(mock_org) + assert pairs == [("lineage_status", "asserted"), ("lineage_status", "unknown"), ("cursor", "page2")] + + @pytest.mark.parametrize("receipt", [None, [], ["inferred"], "unknown"]) + def test_lineage_filter_refuses_unacknowledged_result(self, cs, mock_org, receipt): + mock_org.client.request.return_value = {"images": [], "applied_lineage_status": receipt} + with pytest.raises(RuntimeError, match="not acknowledged"): + cs.list_container_images(lineage_status=["unknown"]) + + @pytest.mark.parametrize("selector", [[], [""], ["verfied"], "unknown", [None]]) + def test_invalid_lineage_filter_never_widens_request(self, cs, mock_org, selector): + with pytest.raises(ValueError, match="lineage_status"): + cs.list_container_images(lineage_status=selector) + mock_org.client.request.assert_not_called() + + @pytest.mark.parametrize("value", [True, False]) + def test_lineage_facets_preserve_explicit_boolean(self, cs, mock_org, value): + cs.get_image_repo_facets(lineage_facet=value, provider=["aws"]) + _, pairs = _get_call(mock_org) + assert pairs == [("provider", "aws"), ("lineage_facet", str(value).lower())] + + def test_caasm_posture_empty_is_unreported_not_unfiltered(self, cs, mock_org): + cs.list_caasm_assets(kind=["device"], source=["okta", "ms_graph"], + posture_encryption=["", "encrypted"], posture_managed=["managed"], + posture_screen_lock=["enabled"], posture_compromised=["false"], + sort="last_seen", cursor="c2", limit=10) + _, pairs = _get_call(mock_org) + assert pairs == [("kind", "device"), ("source", "okta"), ("source", "ms_graph"), + ("posture_encryption", ""), ("posture_encryption", "encrypted"), + ("posture_screen_lock", "enabled"), ("posture_compromised", "false"), + ("posture_managed", "managed"), ("sort", "last_seen"), + ("cursor", "c2"), ("limit", "10")] + + @pytest.mark.parametrize("method", ["export_findings_csv", "export_inventory_csv"]) + def test_chunked_csv_preserves_resume_cursor_and_filter(self, cs, mock_org, method): + mock_org.client.request.return_value = "name\nexample\n# next_cursor=c3\n" + result = getattr(cs, method)(q="example", max_rows=2000, cursor="c2") + _, pairs = _get_call(mock_org) + assert pairs == [("q", "example"), ("max_rows", "2000"), ("cursor", "c2"), ("format", "csv")] + assert result.endswith("# next_cursor=c3\n") + assert mock_org.client.request.call_args.kwargs["raw_response"] is True + + @pytest.mark.parametrize("method", ["export_findings_csv", "export_inventory_csv"]) + @pytest.mark.parametrize("args", [{"max_rows": 0}, {"max_rows": 100001}, {"max_rows": True}, + {"max_rows": 1.5}, {"cursor": "c2"}]) + def test_invalid_export_chunk_never_resumes_at_wrong_position(self, cs, mock_org, method, args): + with pytest.raises(ValueError): + getattr(cs, method)(**args) + mock_org.client.request.assert_not_called() + + +@pytest.mark.parametrize("provider", ["gitlab", "bitbucket"]) +def test_scm_pr_check_resolves_base_without_caller_sha(cs, mock_org, provider): + cs.check_pull_request("acme/api", 42, head_sha="b" * 40, action="synchronize", provider=provider) + _, body = _post_call(mock_org) + assert "base_sha" not in body + assert body["head_sha"] == "b" * 40 + assert body["provider"] == provider + + +def test_github_pr_check_requires_base_before_http(cs, mock_org): + with pytest.raises(ValueError, match="base_sha"): + cs.check_pull_request("acme/api", 42, head_sha="b" * 40, action="opened") + mock_org.client.request.assert_not_called()