Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions package.json
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,8 @@
"preview": "vite preview",
"build-thumbnails": "tsx scripts/build-thumbnails.mts",
"warm-cache": "tsx scripts/warm-cache.mts",
"warm-sample-details": "tsx scripts/warm-sample-details.mts",
"check-query-snapshots": "tsx scripts/check-query-snapshots.mts",
"check-cache-key": "tsx scripts/check-cache-key.mts",
"check-step-labels": "tsx scripts/check-step-labels.mts",
"check-trace-direction": "tsx scripts/check-trace-direction.mts",
Expand Down
1 change: 1 addition & 0 deletions public/thumbs/facilities-upstream-pfos-york-cumberland.svg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
1 change: 1 addition & 0 deletions public/thumbs/samples-downstream-airports-indiana.svg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
1 change: 1 addition & 0 deletions public/thumbs/samples-near-airports-indiana.svg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
1 change: 1 addition & 0 deletions public/thumbs/waterbodies-near-airports-indiana.svg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
19 changes: 18 additions & 1 deletion scripts/check-cache-key.mts
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@
// too-loose means two different questions share an answer and the map shows data
// for a question nobody asked. Hence a runnable check rather than a comment.
import assert from 'node:assert/strict';
import { cacheKey } from '../src/engine/cacheKey';
import { cacheKey, sampleDetailKey } from '../src/engine/cacheKey';
import { PREBUILT_QUERIES } from '../src/constants/prebuiltQueries';
import type { AnalysisQuestion } from '../src/types/query';

Expand Down Expand Up @@ -101,4 +101,21 @@ assert.equal(new Set(keys).size, keys.length, 'dashboard questions must have dis
assert.equal(await cacheKey(PREBUILT_QUERIES[0].question), keys[0], 'key must be stable across calls');
checks += 2;

// Sample-point popup keys (warm-sample-details.mts writes these). The filter
// argument is part of the key: the same point under a substance filter is a
// different observation table.
const spA = 'http://w3id.org/sawgraph/v1/me-egad#d.SamplePoint.1';
const spB = 'http://w3id.org/sawgraph/v1/me-egad#d.SamplePoint.2';
const pfos = { substances: ['http://w3id.org/DSSTox/v1/DTXSID3031864'] };
assert.equal(await sampleDetailKey(spA), await sampleDetailKey(spA), 'sample key must be stable');
assert.notEqual(await sampleDetailKey(spA), await sampleDetailKey(spB), 'different points differ');
assert.notEqual(await sampleDetailKey(spA), await sampleDetailKey(spA, pfos), 'filters are part of the key');
assert.equal(
await sampleDetailKey(spA, { substances: [] }),
await sampleDetailKey(spA),
'an emptied filter must not miss the cache',
);
assert.ok((await sampleDetailKey(spA)).startsWith('s:'), 'sample keys live in the s: namespace');
checks += 5;

console.log(`cache key: ${checks} checks passed`);
15 changes: 14 additions & 1 deletion scripts/check-query-joins.mts
Original file line number Diff line number Diff line change
Expand Up @@ -311,7 +311,20 @@ for (const targetType of ENTITY_TYPES) {
// facilities either way, with the reordered form no slower (13s and 2s against
// 16s and 3s). It reorders because the rule says the state-scoped side is the
// narrower one, and the measurement agrees.
const EXPECTED_REORDER = new Set(['samples-downstream-waste-indiana']);
//
// `facilities-upstream-pfos-york-cumberland` is the same shape the other way
// round: facilities are unconstrained on block A, and block C is PFOS samples
// in two Maine counties. Leading with the samples is rule 1 working, not a
// regression.
//
// `samples-downstream-airports-indiana` is the same shape as the waste one:
// Indiana samples on block A, NAICS 488119/481111 with no region on block C.
// It started reordering when the ranking landed on development.
const EXPECTED_REORDER = new Set([
'samples-downstream-waste-indiana',
'samples-downstream-airports-indiana',
'facilities-upstream-pfos-york-cumberland',
]);

for (const prebuilt of PREBUILT_QUERIES) {
const question: AnalysisQuestion = prebuilt.question;
Expand Down
106 changes: 106 additions & 0 deletions scripts/warm-sample-details.mts
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
// Warms the popup observation rows for the demo prebuilt questions, so a demo
// does not depend on the SPARQL endpoints answering while someone clicks.
//
// CACHE_WRITE_TOKEN=... API_BASE=https://sawgraph-explorer-api-development.up.railway.app \
// npx tsx scripts/warm-sample-details.mts [id-substring ...]
//
// Same trusted-writer path as warm-cache.mts: the real query runs here and only
// the built popup payload is uploaded, under the `s:` key namespace.
//
// ponytail: sequential batches, no concurrency. It is a manual pre-demo run.
import { planPipeline } from '../src/engine/planner';
import { executePipeline } from '../src/engine/executor';
import { sampleDetailKey } from '../src/engine/cacheKey';
import { buildSampleDetailsByIri } from '../src/engine/templates/hydrate';
import { buildSamplePointDetail } from '../src/engine/resultTransformer';
import { executeSparql } from '../src/engine/sparqlClient';
import { PREBUILT_QUERIES } from '../src/constants/prebuiltQueries';
import type { SampleFilters } from '../src/types/query';

const API_BASE = (process.env.API_BASE ?? 'http://localhost:3001').replace(/\/$/, '');
const TOKEN = process.env.CACHE_WRITE_TOKEN;
const BATCH = Number(process.env.BATCH ?? 20);
const MAX_POINTS = Number(process.env.MAX_POINTS ?? 2000);
// Which prebuilts to warm. Defaults to the demo set at the top of the list.
const DEFAULT_IDS = [
'samples-near-airports-indiana',
'facilities-upstream-pfos-york-cumberland',
'samples-downstream-airports-indiana',
];
const ids = process.argv.slice(2).length ? process.argv.slice(2) : DEFAULT_IDS;

if (!TOKEN) {
console.error('CACHE_WRITE_TOKEN is required (must match the API service).');
process.exit(1);
}

const targets = PREBUILT_QUERIES.filter((q) => ids.some((id) => q.id.includes(id)));
if (targets.length === 0) {
console.error(`No prebuilt matched ${ids.join(', ')}`);
process.exit(1);
}

let stored = 0;
let failed = 0;

for (const prebuilt of targets) {
console.log(`\n${prebuilt.title}`);
const filters: SampleFilters | undefined =
prebuilt.question.blockA.type === 'samples'
? prebuilt.question.blockA.sampleFilters
: prebuilt.question.blockC.sampleFilters;

const result = await executePipeline(planPipeline(prebuilt.question), prebuilt.question, () => {});
if (result.status !== 'success') {
console.log(` pipeline ${result.status}, skipping`);
failed++;
continue;
}

const iris = [
...new Set(
Object.values(result.data)
.flat()
.map((row) => row.sp)
.filter((sp): sp is string => Boolean(sp)),
),
].slice(0, MAX_POINTS);
console.log(` ${iris.length} sample point(s)`);

for (let i = 0; i < iris.length; i += BATCH) {
const chunk = iris.slice(i, i + BATCH);
try {
const rows = await executeSparql('federation', buildSampleDetailsByIri(chunk, filters));
const bySp = new Map<string, typeof rows>();
for (const row of rows) {
if (!row.sp) continue;
const arr = bySp.get(row.sp);
if (arr) arr.push(row);
else bySp.set(row.sp, [row]);
}

for (const [sp, spRows] of bySp) {
const detail = buildSamplePointDetail(spRows);
if (!detail) continue;
const res = await fetch(`${API_BASE}/api/results/${await sampleDetailKey(sp, filters)}`, {
method: 'PUT',
headers: { 'Content-Type': 'application/json', 'X-Cache-Token': TOKEN },
body: JSON.stringify({ question: { samplePointIri: sp, filters: filters ?? null }, result: detail }),
});
if (res.ok) stored++;
else {
failed++;
console.log(` upload failed for ${sp} (${res.status})`);
}
}
} catch (err) {
failed++;
console.log(` batch ${i} failed: ${err instanceof Error ? err.message.slice(0, 70) : String(err)}`);
}
process.stdout.write(`\r ${Math.min(i + BATCH, iris.length)}/${iris.length} fetched, ${stored} cached`);
}
process.stdout.write('\n');
}

console.log(`\n${stored} sample point(s) cached, ${failed} failure(s)`);
process.exit(stored === 0 ? 1 : 0);
6 changes: 4 additions & 2 deletions server/src/resultCache.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,9 +3,11 @@ import { pool } from './db.js';

// Payload budget for a single cached result. Measured pipelines run 3-7MB of
// JSON once the internal IRI lists are trimmed client-side; the largest observed
// (a statewide downstream trace with 15MB of river geometry) reached 21MB.
// (a statewide downstream trace with 15MB of river geometry) reached 21MB, and
// the Indiana water-bodies-near-airports demo question 26.6MB of polygons.
// Anything past this is skipped rather than stored — publishing still succeeds.
export const MAX_RESULT_BYTES = 25 * 1024 * 1024;
// Stored gzipped, so the row itself is a fraction of this.
export const MAX_RESULT_BYTES = 32 * 1024 * 1024;

// A clean result is stable until the graph reloads. A partial one is not: it
// means slices failed, and a warmer engine may answer them, so it must not be
Expand Down
5 changes: 3 additions & 2 deletions server/src/routes/results.ts
Original file line number Diff line number Diff line change
Expand Up @@ -43,8 +43,9 @@ resultsRouter.put('/:key', largeJson, async (req: Request, res: Response) => {
if (!body?.question || !body?.result) {
return res.status(400).json({ error: 'question and result are required' });
}
if (!key.startsWith('q:')) {
return res.status(400).json({ error: 'key must be in the q: namespace' });
// q: a whole pipeline result, s: one sample point's popup rows.
if (!key.startsWith('q:') && !key.startsWith('s:')) {
return res.status(400).json({ error: 'key must be in the q: or s: namespace' });
}

const serialized = JSON.stringify(body.result);
Expand Down
20 changes: 19 additions & 1 deletion src/api/resultCacheClient.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,8 @@
import type { PipelineSuccess } from '../engine/executor';
import { cacheKey } from '../engine/cacheKey';
import { cacheKey, sampleDetailKey } from '../engine/cacheKey';
import { fromWire, isWireResult, toWire, type WireResult } from '../engine/wire';
import type { AnalysisQuestion } from '../types/query';
import type { SamplePointDetail } from '../types/map';

function getApiBase(): string {
const base = import.meta.env.VITE_API_BASE_URL;
Expand Down Expand Up @@ -76,3 +77,20 @@ export async function savePublishedResult(
return false;
}
}

// One sample point's popup rows, if a prewarm run stored them. Any failure is
// a miss: the caller then queries the endpoint live, as it always did.
export async function fetchCachedSampleDetail(
samplePointIri: string,
filters?: unknown,
): Promise<SamplePointDetail | null> {
if (cacheDisabled()) return null;
try {
const key = await sampleDetailKey(samplePointIri, filters);
const res = await fetch(`${getApiBase()}/api/results/${key}`);
if (!res.ok) return null;
return (await res.json()) as SamplePointDetail;
} catch {
return null;
}
}
16 changes: 16 additions & 0 deletions src/components/Map/MapPopup.tsx
Original file line number Diff line number Diff line change
@@ -1,3 +1,6 @@
import { useEffect } from 'react';
import { useMap } from 'react-leaflet';
import { Popup as LeafletPopup } from 'leaflet';
import type { MapFeature, SamplePointDetail, SampleRecord } from '../../types/map';
import { useSampleDetails } from '../../hooks/useSampleDetails';
import { useQueryStore } from '../../store/queryStore';
Expand All @@ -18,6 +21,19 @@ function SamplePopup({ feature, isOpen }: { feature: MapFeature; isOpen: boolean
: question.blockC.sampleFilters;
const { data, isLoading, isError } = useSampleDetails(feature.id, isOpen, sampleFilters);
const props = feature.properties;
const map = useMap();

// Leaflet measures a popup once, when it opens (react-leaflet calls
// instance.update() on popupopen and never again). Our rows arrive after
// that, so the first open is sized and anchored for the placeholder and the
// real content spills out of the box. Opening it a second time looks fine
// only because React Query already has the rows. Re-measure when they land.
useEffect(() => {
if (!data) return;
map.eachLayer((layer) => {
if (layer instanceof LeafletPopup) layer.update();
});
}, [data, map]);

if (data) {
return (
Expand Down
91 changes: 91 additions & 0 deletions src/constants/prebuiltQueries.ts
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,97 @@ export interface PrebuiltQuery {
// the thing being asked about — rather than four variations on "samples near
// facilities", which is what a reader took to be the whole tool.
export const PREBUILT_QUERIES: PrebuiltQuery[] = [
// Demo set (Proto-OKN close-out): pinned to the top so they are the four
// cards on the landing page. Their results and sample popups are prewarmed.
// Remove this block after the demo to restore the ordering described above.
{
id: 'samples-near-airports-indiana',
title: 'Samples Near Airport Facilities in Indiana',
description:
'Find PFAS sample points within roughly two miles of Other Airport Operations (NAICS 488119) and Scheduled Passenger Air Transportation (NAICS 481111) facilities across Indiana.',
tags: ['Samples', 'Facilities', 'Near', 'Indiana', 'Airports'],
question: {
blockA: { type: 'samples', region: { stateCode: '18' } },
relationship: { type: 'near', hops: 2 },
blockC: {
type: 'facilities',
facilityFilters: {
industryCodes: ['488119', '481111'],
industryLabels: {
'488119': 'Other Airport Operations',
'481111': 'Scheduled Passenger Air Transportation',
},
},
},
},
},
{
id: 'waterbodies-near-airports-indiana',
title: 'Surface Water Bodies Near Airport Facilities in Indiana',
description:
'Locate surface water bodies within roughly a mile of Other Airport Operations (NAICS 488119) and Scheduled Passenger Air Transportation (NAICS 481111) facilities across Indiana.',
tags: ['Surface Water Bodies', 'Facilities', 'Near', 'Indiana', 'Airports'],
question: {
blockA: { type: 'waterBodies', region: { stateCode: '18' } },
relationship: { type: 'near', hops: 1 },
blockC: {
type: 'facilities',
facilityFilters: {
industryCodes: ['488119', '481111'],
industryLabels: {
'488119': 'Other Airport Operations',
'481111': 'Scheduled Passenger Air Transportation',
},
},
},
},
},
{
id: 'facilities-upstream-pfos-york-cumberland',
title: 'Facilities Upstream from PFOS Samples in York and Cumberland Counties',
description:
'Given samples that tested positive for PFOS in York and Cumberland counties, Maine, trace the river network back the other way: which facilities sit upstream of them, and are therefore candidate sources.',
tags: ['Facilities', 'Samples', 'Upstream', 'PFOS', 'Maine', 'York', 'Cumberland'],
question: {
// Facilities are left unconstrained: the counties scope the samples, and
// an upstream source can sit outside the county the sample is in.
blockA: { type: 'facilities' },
relationship: { type: 'upstream' },
blockC: {
type: 'samples',
region: {
stateCode: '23',
countyCodes: ['23031', '23005'],
countyLabels: { '23031': 'York County, Maine', '23005': 'Cumberland County, Maine' },
},
sampleFilters: {
substances: ['http://w3id.org/DSSTox/v1/DTXSID3031864'],
substanceLabels: { 'http://w3id.org/DSSTox/v1/DTXSID3031864': 'PFOS' },
},
},
},
},
{
id: 'samples-downstream-airports-indiana',
title: 'Indiana Samples Downstream of Airports & Air Transportation Sites',
description:
'Trace downstream flow paths from Other Airport Operations (NAICS 488119) and Scheduled Passenger Air Transportation (NAICS 481111) facilities in Indiana to find PFAS sample points in potentially affected areas.',
tags: ['Samples', 'Facilities', 'Downstream', 'Indiana', 'Airports'],
question: {
blockA: { type: 'samples', region: { stateCode: '18' } },
relationship: { type: 'downstream' },
blockC: {
type: 'facilities',
facilityFilters: {
industryCodes: ['488119', '481111'],
industryLabels: {
'488119': 'Other Airport Operations',
'481111': 'Scheduled Passenger Air Transportation',
},
},
},
},
},
{
id: 'facilities-upstream-pfhpa-gw-cumberland',
title: 'Facilities Upstream from PFHpA Groundwater Contamination in Cumberland County',
Expand Down
13 changes: 13 additions & 0 deletions src/engine/cacheKey.ts
Original file line number Diff line number Diff line change
Expand Up @@ -92,3 +92,16 @@ export async function cacheKey(question: AnalysisQuestion): Promise<string> {
const hex = [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, '0')).join('');
return `q:${hex}`;
}

// Key for one sample point's observation table (the popup rows), cached
// server-side so a demo does not depend on the SPARQL endpoints answering on
// click. Same versioning as the question key: a graph reload retires both.
export async function sampleDetailKey(
samplePointIri: string,
filters?: unknown,
): Promise<string> {
const body = `${DATA_VERSION}|${samplePointIri}|${JSON.stringify(canonical(filters ?? {}) ?? {})}`;
const digest = await crypto.subtle.digest('SHA-256', new TextEncoder().encode(body));
const hex = [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, '0')).join('');
return `s:${hex}`;
}
Loading
Loading