diff --git a/README.md b/README.md index dc94418..cf7a81e 100644 --- a/README.md +++ b/README.md @@ -63,6 +63,16 @@ the read and write keys are kept separate on purpose so the browser-facing read **populating the index.** backfill and refresh now feed the sink, so `pnpm backfill` fills meili as it walks each user's pds. on an existing deployment the event records are usually already in d1, so `pnpm meili:reindex` is faster: it replays the stored `community.lexicon.calendar.event` rows straight from d1 into the index with no network walk and no d1 writes. both paths apply the same discoverable filter as live ingest, and the sink applies the index settings on its first write, so a fresh index gets the right filterable fields and re-running either is idempotent. `pnpm meili:reindex:remote` targets the deployed d1 and needs the same wrangler `env.production` that `pnpm backfill:remote` uses. +### Near-me geocoding (optional) + +Many events carry only a street address, no coordinates — so they never surface in near-me, which filters on the Meilisearch document's `_geo`. This resolves those addresses to coordinates and writes `_geo` back into the same index, making address-only events near-me-visible. It layers on top of the sink above: no extra service — it rides the existing cron and writes the same index. Leave it untouched and it runs keyless against public [Nominatim](https://nominatim.org/) at a safe trickle; until an address resolves, that event simply stays out of near-me. + +**How it runs.** The cron already calls a geocode "drip" every minute; it self-throttles to once per ~30 min via a D1 marker, resolves up to 50 new addresses per run (25 on public Nominatim), and `PATCH`es `_geo` into Meilisearch. Every result is cached — including negative results, so an ungeocodable address isn't retried every run. There is nothing to set up: like the app's other D1 tables, the geocode cache and its cadence marker are defined in code and self-heal on first run. The drip no-ops entirely until the **write sink** above is configured, so enabling search is the only switch. + +**Picking a geocoder.** The default is keyless public OSM Nominatim — fine for the steady-state drip's low volume. Set `GEOCODER_USER_AGENT` to a string identifying your deployment (Nominatim's [usage policy](https://operations.osmfoundation.org/policies/nominatim/) requires a real contact; on the public host the per-run cap is held to 25 and the throttle floored to ≥1 req/s). For real volume — and for the bulk backfill below — use [LocationIQ](https://locationiq.com/) (an API-compatible hosted Nominatim): set `GEOCODER_URL=https://us1.locationiq.com/v1/search` and `GEOCODER_KEY`, which lifts the per-run cap to 50 and honors your `GEOCODE_SLEEP_MS` (minimum ms between calls, default 1100). A key with an unset or public `GEOCODER_URL` is *ignored* — you stay on public Nominatim — so always set the URL too. + +**Backfilling an existing corpus.** The drip only trickles, so to resolve a backlog run the off-Cloudflare CLI against the deployed D1: `pnpm -C apps/web geocode:backfill --limit 50`. It reaches D1 over the REST API, so it needs `CLOUDFLARE_ACCOUNT_ID` / `CLOUDFLARE_API_TOKEN` / `D1_DATABASE_ID` and `MEILI_URL` / `MEILI_KEY` (plus `SEARCH_INDEX` if not `events`), and a LocationIQ `GEOCODER_URL` / `GEOCODER_KEY`. It refuses a bulk or uncapped run against public Nominatim (keyless is capped to `--limit 1..25`). Useful flags: `--limit N` (`0` = no cap), `--dry-run`, `--retry-negative` (re-attempt negatively-cached addresses), and `--allow-public-nominatim` (override the public-host guard). Like search itself, geocoding only helps once the sink is feeding the index, so run this after the rollout steps above. + ## contributing open for contributions by all :) diff --git a/apps/web/.env.example b/apps/web/.env.example index 6d816d2..7e99bf0 100644 --- a/apps/web/.env.example +++ b/apps/web/.env.example @@ -19,3 +19,14 @@ COOKIE_SECRET= # The index is SEARCH_INDEX above (the sink writes the same index it reads). # SEARCH_SINK_URL=http://localhost:7700 # SEARCH_SINK_API_KEY= +# Near-me geocoding (optional; layered on the search sink above). Resolves the +# coordinates of address-only events and writes _geo into the index so they +# appear in near-me. Unset → the in-cron drip runs keyless against public OSM +# Nominatim at a safe trickle. See the "near-me geocoding" section in the README. +# For real volume / a bulk backfill, use LocationIQ (set both URL and KEY): +# GEOCODER_URL=https://us1.locationiq.com/v1/search +# GEOCODER_KEY= +# Identify your deployment to public Nominatim (its usage policy requires it): +# GEOCODER_USER_AGENT=atmo-events (you@example.com) +# Minimum ms between geocoder calls (default 1100; floored to 1000 on public Nominatim): +# GEOCODE_SLEEP_MS=1100 diff --git a/apps/web/scripts/geocode-cache.sql b/apps/web/scripts/geocode-cache.sql deleted file mode 100644 index 20c5f23..0000000 --- a/apps/web/scripts/geocode-cache.sql +++ /dev/null @@ -1,14 +0,0 @@ --- Derived-coordinate cache for address-only events + the geocode job's worklist --- and done-marker (Meili can't filter "missing _geo", so we track resolution --- ourselves). Idempotent / restartable. Applied once to the openmeet-atmo D1; --- the sink reads it, the external geocode job writes it. -CREATE TABLE IF NOT EXISTS geocode_cache ( - address_norm TEXT PRIMARY KEY, -- normalized address key (address-norm.ts) - lat REAL, -- NULL when unresolved (negative cache) - lng REAL, - precision TEXT, -- provider-reported granularity - source TEXT NOT NULL, -- e.g. 'locationiq' / 'nominatim' - geocoded_at INTEGER NOT NULL, -- epoch ms - fail_count INTEGER NOT NULL DEFAULT 0, - last_error TEXT -); diff --git a/apps/web/src/lib/search/server/geocode-job.ts b/apps/web/src/lib/search/server/geocode-job.ts index 6f70140..5d24389 100644 --- a/apps/web/src/lib/search/server/geocode-job.ts +++ b/apps/web/src/lib/search/server/geocode-job.ts @@ -146,8 +146,9 @@ export async function runGeocodeJob(opts: GeocodeJobOptions): Promise= 0 ? sleepMs : 1100; - // Defensive: the table is created by geocode-cache.sql, but a fresh DB - // shouldn't make the job crash before it can self-heal. + // This CREATE is the authoritative definition of geocode_cache: like the + // app's other D1 tables it lives in code and self-heals, so a fresh DB just + // works on first run with nothing to apply by hand. await d1.query( `CREATE TABLE IF NOT EXISTS geocode_cache ( address_norm TEXT PRIMARY KEY, lat REAL, lng REAL, precision TEXT,