diff --git a/package.json b/package.json index 69af60f..080ffe9 100644 --- a/package.json +++ b/package.json @@ -10,6 +10,8 @@ "preview": "vite preview", "build-thumbnails": "tsx scripts/build-thumbnails.mts", "warm-cache": "tsx scripts/warm-cache.mts", + "warm-sample-details": "tsx scripts/warm-sample-details.mts", + "check-query-snapshots": "tsx scripts/check-query-snapshots.mts", "check-cache-key": "tsx scripts/check-cache-key.mts", "check-step-labels": "tsx scripts/check-step-labels.mts", "check-trace-direction": "tsx scripts/check-trace-direction.mts", diff --git a/public/thumbs/facilities-upstream-pfos-york-cumberland.svg b/public/thumbs/facilities-upstream-pfos-york-cumberland.svg new file mode 100644 index 0000000..4b9f716 --- /dev/null +++ b/public/thumbs/facilities-upstream-pfos-york-cumberland.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/public/thumbs/samples-downstream-airports-indiana.svg b/public/thumbs/samples-downstream-airports-indiana.svg new file mode 100644 index 0000000..31eddd4 --- /dev/null +++ b/public/thumbs/samples-downstream-airports-indiana.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/public/thumbs/samples-near-airports-indiana.svg b/public/thumbs/samples-near-airports-indiana.svg new file mode 100644 index 0000000..0f52b83 --- /dev/null +++ b/public/thumbs/samples-near-airports-indiana.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/public/thumbs/waterbodies-near-airports-indiana.svg b/public/thumbs/waterbodies-near-airports-indiana.svg new file mode 100644 index 0000000..3f4f08c --- /dev/null +++ b/public/thumbs/waterbodies-near-airports-indiana.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/scripts/check-cache-key.mts b/scripts/check-cache-key.mts index 39d7df0..1bf5992 100644 --- a/scripts/check-cache-key.mts +++ b/scripts/check-cache-key.mts @@ -7,7 +7,7 @@ // too-loose means two different questions share an answer and the map shows data // for a question nobody asked. Hence a runnable check rather than a comment. import assert from 'node:assert/strict'; -import { cacheKey } from '../src/engine/cacheKey'; +import { cacheKey, sampleDetailKey } from '../src/engine/cacheKey'; import { PREBUILT_QUERIES } from '../src/constants/prebuiltQueries'; import type { AnalysisQuestion } from '../src/types/query'; @@ -101,4 +101,21 @@ assert.equal(new Set(keys).size, keys.length, 'dashboard questions must have dis assert.equal(await cacheKey(PREBUILT_QUERIES[0].question), keys[0], 'key must be stable across calls'); checks += 2; +// Sample-point popup keys (warm-sample-details.mts writes these). The filter +// argument is part of the key: the same point under a substance filter is a +// different observation table. +const spA = 'http://w3id.org/sawgraph/v1/me-egad#d.SamplePoint.1'; +const spB = 'http://w3id.org/sawgraph/v1/me-egad#d.SamplePoint.2'; +const pfos = { substances: ['http://w3id.org/DSSTox/v1/DTXSID3031864'] }; +assert.equal(await sampleDetailKey(spA), await sampleDetailKey(spA), 'sample key must be stable'); +assert.notEqual(await sampleDetailKey(spA), await sampleDetailKey(spB), 'different points differ'); +assert.notEqual(await sampleDetailKey(spA), await sampleDetailKey(spA, pfos), 'filters are part of the key'); +assert.equal( + await sampleDetailKey(spA, { substances: [] }), + await sampleDetailKey(spA), + 'an emptied filter must not miss the cache', +); +assert.ok((await sampleDetailKey(spA)).startsWith('s:'), 'sample keys live in the s: namespace'); +checks += 5; + console.log(`cache key: ${checks} checks passed`); diff --git a/scripts/check-query-joins.mts b/scripts/check-query-joins.mts index 1d659c6..b8bf432 100644 --- a/scripts/check-query-joins.mts +++ b/scripts/check-query-joins.mts @@ -311,7 +311,20 @@ for (const targetType of ENTITY_TYPES) { // facilities either way, with the reordered form no slower (13s and 2s against // 16s and 3s). It reorders because the rule says the state-scoped side is the // narrower one, and the measurement agrees. -const EXPECTED_REORDER = new Set(['samples-downstream-waste-indiana']); +// +// `facilities-upstream-pfos-york-cumberland` is the same shape the other way +// round: facilities are unconstrained on block A, and block C is PFOS samples +// in two Maine counties. Leading with the samples is rule 1 working, not a +// regression. +// +// `samples-downstream-airports-indiana` is the same shape as the waste one: +// Indiana samples on block A, NAICS 488119/481111 with no region on block C. +// It started reordering when the ranking landed on development. +const EXPECTED_REORDER = new Set([ + 'samples-downstream-waste-indiana', + 'samples-downstream-airports-indiana', + 'facilities-upstream-pfos-york-cumberland', +]); for (const prebuilt of PREBUILT_QUERIES) { const question: AnalysisQuestion = prebuilt.question; diff --git a/scripts/warm-sample-details.mts b/scripts/warm-sample-details.mts new file mode 100644 index 0000000..5ece3a3 --- /dev/null +++ b/scripts/warm-sample-details.mts @@ -0,0 +1,106 @@ +// Warms the popup observation rows for the demo prebuilt questions, so a demo +// does not depend on the SPARQL endpoints answering while someone clicks. +// +// CACHE_WRITE_TOKEN=... API_BASE=https://sawgraph-explorer-api-development.up.railway.app \ +// npx tsx scripts/warm-sample-details.mts [id-substring ...] +// +// Same trusted-writer path as warm-cache.mts: the real query runs here and only +// the built popup payload is uploaded, under the `s:` key namespace. +// +// ponytail: sequential batches, no concurrency. It is a manual pre-demo run. +import { planPipeline } from '../src/engine/planner'; +import { executePipeline } from '../src/engine/executor'; +import { sampleDetailKey } from '../src/engine/cacheKey'; +import { buildSampleDetailsByIri } from '../src/engine/templates/hydrate'; +import { buildSamplePointDetail } from '../src/engine/resultTransformer'; +import { executeSparql } from '../src/engine/sparqlClient'; +import { PREBUILT_QUERIES } from '../src/constants/prebuiltQueries'; +import type { SampleFilters } from '../src/types/query'; + +const API_BASE = (process.env.API_BASE ?? 'http://localhost:3001').replace(/\/$/, ''); +const TOKEN = process.env.CACHE_WRITE_TOKEN; +const BATCH = Number(process.env.BATCH ?? 20); +const MAX_POINTS = Number(process.env.MAX_POINTS ?? 2000); +// Which prebuilts to warm. Defaults to the demo set at the top of the list. +const DEFAULT_IDS = [ + 'samples-near-airports-indiana', + 'facilities-upstream-pfos-york-cumberland', + 'samples-downstream-airports-indiana', +]; +const ids = process.argv.slice(2).length ? process.argv.slice(2) : DEFAULT_IDS; + +if (!TOKEN) { + console.error('CACHE_WRITE_TOKEN is required (must match the API service).'); + process.exit(1); +} + +const targets = PREBUILT_QUERIES.filter((q) => ids.some((id) => q.id.includes(id))); +if (targets.length === 0) { + console.error(`No prebuilt matched ${ids.join(', ')}`); + process.exit(1); +} + +let stored = 0; +let failed = 0; + +for (const prebuilt of targets) { + console.log(`\n${prebuilt.title}`); + const filters: SampleFilters | undefined = + prebuilt.question.blockA.type === 'samples' + ? prebuilt.question.blockA.sampleFilters + : prebuilt.question.blockC.sampleFilters; + + const result = await executePipeline(planPipeline(prebuilt.question), prebuilt.question, () => {}); + if (result.status !== 'success') { + console.log(` pipeline ${result.status}, skipping`); + failed++; + continue; + } + + const iris = [ + ...new Set( + Object.values(result.data) + .flat() + .map((row) => row.sp) + .filter((sp): sp is string => Boolean(sp)), + ), + ].slice(0, MAX_POINTS); + console.log(` ${iris.length} sample point(s)`); + + for (let i = 0; i < iris.length; i += BATCH) { + const chunk = iris.slice(i, i + BATCH); + try { + const rows = await executeSparql('federation', buildSampleDetailsByIri(chunk, filters)); + const bySp = new Map(); + for (const row of rows) { + if (!row.sp) continue; + const arr = bySp.get(row.sp); + if (arr) arr.push(row); + else bySp.set(row.sp, [row]); + } + + for (const [sp, spRows] of bySp) { + const detail = buildSamplePointDetail(spRows); + if (!detail) continue; + const res = await fetch(`${API_BASE}/api/results/${await sampleDetailKey(sp, filters)}`, { + method: 'PUT', + headers: { 'Content-Type': 'application/json', 'X-Cache-Token': TOKEN }, + body: JSON.stringify({ question: { samplePointIri: sp, filters: filters ?? null }, result: detail }), + }); + if (res.ok) stored++; + else { + failed++; + console.log(` upload failed for ${sp} (${res.status})`); + } + } + } catch (err) { + failed++; + console.log(` batch ${i} failed: ${err instanceof Error ? err.message.slice(0, 70) : String(err)}`); + } + process.stdout.write(`\r ${Math.min(i + BATCH, iris.length)}/${iris.length} fetched, ${stored} cached`); + } + process.stdout.write('\n'); +} + +console.log(`\n${stored} sample point(s) cached, ${failed} failure(s)`); +process.exit(stored === 0 ? 1 : 0); diff --git a/server/src/resultCache.ts b/server/src/resultCache.ts index 0e0fbd4..fac1a29 100644 --- a/server/src/resultCache.ts +++ b/server/src/resultCache.ts @@ -3,9 +3,11 @@ import { pool } from './db.js'; // Payload budget for a single cached result. Measured pipelines run 3-7MB of // JSON once the internal IRI lists are trimmed client-side; the largest observed -// (a statewide downstream trace with 15MB of river geometry) reached 21MB. +// (a statewide downstream trace with 15MB of river geometry) reached 21MB, and +// the Indiana water-bodies-near-airports demo question 26.6MB of polygons. // Anything past this is skipped rather than stored — publishing still succeeds. -export const MAX_RESULT_BYTES = 25 * 1024 * 1024; +// Stored gzipped, so the row itself is a fraction of this. +export const MAX_RESULT_BYTES = 32 * 1024 * 1024; // A clean result is stable until the graph reloads. A partial one is not: it // means slices failed, and a warmer engine may answer them, so it must not be diff --git a/server/src/routes/results.ts b/server/src/routes/results.ts index 06aee6e..f2c1f80 100644 --- a/server/src/routes/results.ts +++ b/server/src/routes/results.ts @@ -43,8 +43,9 @@ resultsRouter.put('/:key', largeJson, async (req: Request, res: Response) => { if (!body?.question || !body?.result) { return res.status(400).json({ error: 'question and result are required' }); } - if (!key.startsWith('q:')) { - return res.status(400).json({ error: 'key must be in the q: namespace' }); + // q: a whole pipeline result, s: one sample point's popup rows. + if (!key.startsWith('q:') && !key.startsWith('s:')) { + return res.status(400).json({ error: 'key must be in the q: or s: namespace' }); } const serialized = JSON.stringify(body.result); diff --git a/src/api/resultCacheClient.ts b/src/api/resultCacheClient.ts index 1802dc4..1f59ced 100644 --- a/src/api/resultCacheClient.ts +++ b/src/api/resultCacheClient.ts @@ -1,7 +1,8 @@ import type { PipelineSuccess } from '../engine/executor'; -import { cacheKey } from '../engine/cacheKey'; +import { cacheKey, sampleDetailKey } from '../engine/cacheKey'; import { fromWire, isWireResult, toWire, type WireResult } from '../engine/wire'; import type { AnalysisQuestion } from '../types/query'; +import type { SamplePointDetail } from '../types/map'; function getApiBase(): string { const base = import.meta.env.VITE_API_BASE_URL; @@ -76,3 +77,20 @@ export async function savePublishedResult( return false; } } + +// One sample point's popup rows, if a prewarm run stored them. Any failure is +// a miss: the caller then queries the endpoint live, as it always did. +export async function fetchCachedSampleDetail( + samplePointIri: string, + filters?: unknown, +): Promise { + if (cacheDisabled()) return null; + try { + const key = await sampleDetailKey(samplePointIri, filters); + const res = await fetch(`${getApiBase()}/api/results/${key}`); + if (!res.ok) return null; + return (await res.json()) as SamplePointDetail; + } catch { + return null; + } +} diff --git a/src/components/Map/MapPopup.tsx b/src/components/Map/MapPopup.tsx index 7c48580..a96a594 100644 --- a/src/components/Map/MapPopup.tsx +++ b/src/components/Map/MapPopup.tsx @@ -1,3 +1,6 @@ +import { useEffect } from 'react'; +import { useMap } from 'react-leaflet'; +import { Popup as LeafletPopup } from 'leaflet'; import type { MapFeature, SamplePointDetail, SampleRecord } from '../../types/map'; import { useSampleDetails } from '../../hooks/useSampleDetails'; import { useQueryStore } from '../../store/queryStore'; @@ -18,6 +21,19 @@ function SamplePopup({ feature, isOpen }: { feature: MapFeature; isOpen: boolean : question.blockC.sampleFilters; const { data, isLoading, isError } = useSampleDetails(feature.id, isOpen, sampleFilters); const props = feature.properties; + const map = useMap(); + + // Leaflet measures a popup once, when it opens (react-leaflet calls + // instance.update() on popupopen and never again). Our rows arrive after + // that, so the first open is sized and anchored for the placeholder and the + // real content spills out of the box. Opening it a second time looks fine + // only because React Query already has the rows. Re-measure when they land. + useEffect(() => { + if (!data) return; + map.eachLayer((layer) => { + if (layer instanceof LeafletPopup) layer.update(); + }); + }, [data, map]); if (data) { return ( diff --git a/src/constants/prebuiltQueries.ts b/src/constants/prebuiltQueries.ts index 0178784..d5f2505 100644 --- a/src/constants/prebuiltQueries.ts +++ b/src/constants/prebuiltQueries.ts @@ -14,6 +14,97 @@ export interface PrebuiltQuery { // the thing being asked about — rather than four variations on "samples near // facilities", which is what a reader took to be the whole tool. export const PREBUILT_QUERIES: PrebuiltQuery[] = [ + // Demo set (Proto-OKN close-out): pinned to the top so they are the four + // cards on the landing page. Their results and sample popups are prewarmed. + // Remove this block after the demo to restore the ordering described above. + { + id: 'samples-near-airports-indiana', + title: 'Samples Near Airport Facilities in Indiana', + description: + 'Find PFAS sample points within roughly two miles of Other Airport Operations (NAICS 488119) and Scheduled Passenger Air Transportation (NAICS 481111) facilities across Indiana.', + tags: ['Samples', 'Facilities', 'Near', 'Indiana', 'Airports'], + question: { + blockA: { type: 'samples', region: { stateCode: '18' } }, + relationship: { type: 'near', hops: 2 }, + blockC: { + type: 'facilities', + facilityFilters: { + industryCodes: ['488119', '481111'], + industryLabels: { + '488119': 'Other Airport Operations', + '481111': 'Scheduled Passenger Air Transportation', + }, + }, + }, + }, + }, + { + id: 'waterbodies-near-airports-indiana', + title: 'Surface Water Bodies Near Airport Facilities in Indiana', + description: + 'Locate surface water bodies within roughly a mile of Other Airport Operations (NAICS 488119) and Scheduled Passenger Air Transportation (NAICS 481111) facilities across Indiana.', + tags: ['Surface Water Bodies', 'Facilities', 'Near', 'Indiana', 'Airports'], + question: { + blockA: { type: 'waterBodies', region: { stateCode: '18' } }, + relationship: { type: 'near', hops: 1 }, + blockC: { + type: 'facilities', + facilityFilters: { + industryCodes: ['488119', '481111'], + industryLabels: { + '488119': 'Other Airport Operations', + '481111': 'Scheduled Passenger Air Transportation', + }, + }, + }, + }, + }, + { + id: 'facilities-upstream-pfos-york-cumberland', + title: 'Facilities Upstream from PFOS Samples in York and Cumberland Counties', + description: + 'Given samples that tested positive for PFOS in York and Cumberland counties, Maine, trace the river network back the other way: which facilities sit upstream of them, and are therefore candidate sources.', + tags: ['Facilities', 'Samples', 'Upstream', 'PFOS', 'Maine', 'York', 'Cumberland'], + question: { + // Facilities are left unconstrained: the counties scope the samples, and + // an upstream source can sit outside the county the sample is in. + blockA: { type: 'facilities' }, + relationship: { type: 'upstream' }, + blockC: { + type: 'samples', + region: { + stateCode: '23', + countyCodes: ['23031', '23005'], + countyLabels: { '23031': 'York County, Maine', '23005': 'Cumberland County, Maine' }, + }, + sampleFilters: { + substances: ['http://w3id.org/DSSTox/v1/DTXSID3031864'], + substanceLabels: { 'http://w3id.org/DSSTox/v1/DTXSID3031864': 'PFOS' }, + }, + }, + }, + }, + { + id: 'samples-downstream-airports-indiana', + title: 'Indiana Samples Downstream of Airports & Air Transportation Sites', + description: + 'Trace downstream flow paths from Other Airport Operations (NAICS 488119) and Scheduled Passenger Air Transportation (NAICS 481111) facilities in Indiana to find PFAS sample points in potentially affected areas.', + tags: ['Samples', 'Facilities', 'Downstream', 'Indiana', 'Airports'], + question: { + blockA: { type: 'samples', region: { stateCode: '18' } }, + relationship: { type: 'downstream' }, + blockC: { + type: 'facilities', + facilityFilters: { + industryCodes: ['488119', '481111'], + industryLabels: { + '488119': 'Other Airport Operations', + '481111': 'Scheduled Passenger Air Transportation', + }, + }, + }, + }, + }, { id: 'facilities-upstream-pfhpa-gw-cumberland', title: 'Facilities Upstream from PFHpA Groundwater Contamination in Cumberland County', diff --git a/src/engine/cacheKey.ts b/src/engine/cacheKey.ts index 935b056..fb31917 100644 --- a/src/engine/cacheKey.ts +++ b/src/engine/cacheKey.ts @@ -92,3 +92,16 @@ export async function cacheKey(question: AnalysisQuestion): Promise { const hex = [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, '0')).join(''); return `q:${hex}`; } + +// Key for one sample point's observation table (the popup rows), cached +// server-side so a demo does not depend on the SPARQL endpoints answering on +// click. Same versioning as the question key: a graph reload retires both. +export async function sampleDetailKey( + samplePointIri: string, + filters?: unknown, +): Promise { + const body = `${DATA_VERSION}|${samplePointIri}|${JSON.stringify(canonical(filters ?? {}) ?? {})}`; + const digest = await crypto.subtle.digest('SHA-256', new TextEncoder().encode(body)); + const hex = [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, '0')).join(''); + return `s:${hex}`; +} diff --git a/src/hooks/useSampleDetails.ts b/src/hooks/useSampleDetails.ts index 1acd855..27558ed 100644 --- a/src/hooks/useSampleDetails.ts +++ b/src/hooks/useSampleDetails.ts @@ -2,6 +2,7 @@ import { useQuery } from '@tanstack/react-query'; import { executeSparql } from '../engine/sparqlClient'; import { buildSampleDetailsByIri } from '../engine/templates/hydrate'; import { buildSamplePointDetail } from '../engine/resultTransformer'; +import { fetchCachedSampleDetail } from '../api/resultCacheClient'; import type { SamplePointDetail } from '../types/map'; import type { SampleFilters } from '../types/query'; @@ -11,6 +12,9 @@ import type { SampleFilters } from '../types/query'; // (37,425 rows for one Maine question) to fill popups that open one at a time. // One sample point is ~30KB. `enabled` is driven by the popup's open state so // nothing is fetched until a user actually asks for it. +// +// Prewarmed popups (scripts/warm-sample-details.mts) are served from the API +// cache; everything else falls through to the live endpoint as before. export function useSampleDetails( samplePointIri: string | null, enabled: boolean, @@ -22,10 +26,11 @@ export function useSampleDetails( staleTime: Infinity, retry: false, queryFn: async () => { - const rows = await executeSparql( - 'federation', - buildSampleDetailsByIri([samplePointIri as string], filters), - ); + const iri = samplePointIri as string; + const cached = await fetchCachedSampleDetail(iri, filters); + if (cached) return cached; + + const rows = await executeSparql('federation', buildSampleDetailsByIri([iri], filters)); return buildSamplePointDetail(rows); }, });