diff --git a/.github/workflows/container-palomar-aws.yaml b/.github/workflows/container-palomar-aws.yaml deleted file mode 100644 index 6c7cb456..00000000 --- a/.github/workflows/container-palomar-aws.yaml +++ /dev/null @@ -1,52 +0,0 @@ -name: container-palomar-aws -on: [push] -env: - REGISTRY: ${{ secrets.AWS_ECR_REGISTRY_USEAST2_PACKAGES_REGISTRY }} - USERNAME: ${{ secrets.AWS_ECR_REGISTRY_USEAST2_PACKAGES_USERNAME }} - PASSWORD: ${{ secrets.AWS_ECR_REGISTRY_USEAST2_PACKAGES_PASSWORD }} - # github.repository as / - IMAGE_NAME: palomar - -jobs: - container-palomar-aws: - if: github.repository == 'bluesky-social/indigo' - runs-on: ubuntu-latest - permissions: - contents: read - packages: write - id-token: write - - steps: - - name: Checkout repository - uses: actions/checkout@v3 - - - name: Setup Docker buildx - uses: docker/setup-buildx-action@v1 - - - name: Log into registry ${{ env.REGISTRY }} - uses: docker/login-action@v2 - with: - registry: ${{ env.REGISTRY }} - username: ${{ env.USERNAME }} - password: ${{ env.PASSWORD }} - - - name: Extract Docker metadata - id: meta - uses: docker/metadata-action@v4 - with: - images: | - ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }} - tags: | - type=sha,enable=true,priority=100,prefix=,suffix=,format=long - - - name: Build and push Docker image - id: build-and-push - uses: docker/build-push-action@v4 - with: - context: . - file: ./cmd/palomar/Dockerfile - push: ${{ github.event_name != 'pull_request' }} - tags: ${{ steps.meta.outputs.tags }} - labels: ${{ steps.meta.outputs.labels }} - cache-from: type=gha - cache-to: type=gha,mode=max diff --git a/.github/workflows/container-palomar-ghcr.yaml b/.github/workflows/container-palomar-ghcr.yaml deleted file mode 100644 index 7b6df603..00000000 --- a/.github/workflows/container-palomar-ghcr.yaml +++ /dev/null @@ -1,53 +0,0 @@ -name: container-palomar-ghcr -on: [push] -env: - REGISTRY: ghcr.io - USERNAME: ${{ github.actor }} - PASSWORD: ${{ secrets.GITHUB_TOKEN }} - - # github.repository as / - IMAGE_NAME: ${{ github.repository }} - -jobs: - container-palomar-ghcr: - if: github.repository == 'bluesky-social/indigo' - runs-on: ubuntu-latest - permissions: - contents: read - packages: write - id-token: write - - steps: - - name: Checkout repository - uses: actions/checkout@v3 - - - name: Setup Docker buildx - uses: docker/setup-buildx-action@v1 - - - name: Log into registry ${{ env.REGISTRY }} - uses: docker/login-action@v2 - with: - registry: ${{ env.REGISTRY }} - username: ${{ env.USERNAME }} - password: ${{ env.PASSWORD }} - - - name: Extract Docker metadata - id: meta - uses: docker/metadata-action@v4 - with: - images: | - ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }} - tags: | - type=sha,enable=true,priority=100,prefix=palomar:,suffix=,format=long - - - name: Build and push Docker image - id: build-and-push - uses: docker/build-push-action@v4 - with: - context: . - file: ./cmd/palomar/Dockerfile - push: ${{ github.event_name != 'pull_request' }} - tags: ${{ steps.meta.outputs.tags }} - labels: ${{ steps.meta.outputs.labels }} - cache-from: type=gha - cache-to: type=gha,mode=max diff --git a/.gitignore b/.gitignore index 3ba7a279..2769fd6a 100644 --- a/.gitignore +++ b/.gitignore @@ -29,7 +29,6 @@ test-coverage.out /gosky /hepa /lexgen -/palomar /rainbow /relay /sonar diff --git a/HACKING.md b/HACKING.md index 7ed3ef1f..c4a4e4c5 100644 --- a/HACKING.md +++ b/HACKING.md @@ -4,7 +4,6 @@ Run with, eg, `go run ./cmd/rainbow`): - `cmd/relay`: new (sync v1.1) relay daemon -- `cmd/palomar`: search indexer and query service (OpenSearch) - `cmd/gosky`: client CLI for talking to a PDS - `cmd/lexgen`: codegen tool for lexicons (Lexicon JSON to Go package) - `cmd/stress`: connects to local/default PDS and creates a ton of random posts diff --git a/Makefile b/Makefile index e6abbecf..3aaa933f 100644 --- a/Makefile +++ b/Makefile @@ -24,7 +24,6 @@ build: ## Build all executables go build ./cmd/fakermaker go build ./cmd/hepa go build -o ./sonar-cli ./cmd/sonar - go build ./cmd/palomar go build ./cmd/tap .PHONY: all @@ -79,11 +78,6 @@ cborgen: ## Run codegen tool for CBOR serialization .env: if [ ! -f ".env" ]; then cp example.dev.env .env; fi -.PHONY: run-dev-opensearch -run-dev-opensearch: .env ## Runs a local opensearch instance - docker build -f cmd/palomar/Dockerfile.opensearch . -t opensearch-palomar - docker run -p 9200:9200 -p 9600:9600 -e "discovery.type=single-node" -e "plugins.security.disabled=true" -e "OPENSEARCH_INITIAL_ADMIN_PASSWORD=0penSearch-Pal0mar" opensearch-palomar - .PHONY: run-dev-relay run-dev-relay: .env ## Runs relay for local dev LOG_LEVEL=info go run ./cmd/relay --admin-password localdev serve @@ -107,10 +101,6 @@ run-relay-image: docker run -p 2470:2470 relay /relay serve --admin-password localdev # --crawl-insecure-ws -.PHONY: run-dev-search -run-dev-search: .env ## Runs search daemon for local dev - GOLOG_LOG_LEVEL=info go run ./cmd/palomar run - .PHONY: sonar-up sonar-up: # Runs sonar docker container docker compose -f cmd/sonar/docker-compose.yml up --build -d || docker-compose -f cmd/sonar/docker-compose.yml up --build -d diff --git a/README.md b/README.md index 87f9ec8e..086910f1 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,6 @@ Go will fetch dependencies, compile, and install `tap` or another service with a - **relay** ([README](./cmd/relay/README.md)): relay reference implementation - **rainbow** ([README](./cmd/rainbow/README.md)): firehose "splitter" or "fan-out" service - **hepa** ([README](./cmd/hepa/README.md)): auto-moderation bot for [Ozone](https://ozone.tools) -- **palomar** ([README](./cmd/palomar/README.md)): fulltext search service for **Developer Tools:** diff --git a/cmd/palomar/Dockerfile b/cmd/palomar/Dockerfile deleted file mode 100644 index 362eb00e..00000000 --- a/cmd/palomar/Dockerfile +++ /dev/null @@ -1,37 +0,0 @@ -# Run this dockerfile from the top level of the indigo git repository like: -# -# podman build -f ./cmd/palomar/Dockerfile -t palomar . - -### Compile stage -FROM golang:1.26-alpine3.22 AS build-env -RUN apk add --no-cache build-base make git - -ADD . /dockerbuild -WORKDIR /dockerbuild - -# timezone data for alpine builds -ENV GOEXPERIMENT=loopvar -RUN GIT_VERSION=$(git describe --tags --long --always) && \ - go build -tags timetzdata -o /palomar ./cmd/palomar - -### Run stage -FROM alpine:3.22 - -RUN apk add --no-cache --update dumb-init ca-certificates runit -ENTRYPOINT ["dumb-init", "--"] - -WORKDIR / -RUN mkdir -p data/palomar -COPY --from=build-env /palomar / - -# small things to make golang binaries work well under alpine -ENV GODEBUG=netdns=go -ENV TZ=Etc/UTC - -EXPOSE 3999 - -CMD ["/palomar", "run"] - -LABEL org.opencontainers.image.source=https://github.com/bluesky-social/indigo -LABEL org.opencontainers.image.description="atproto Search Service (for app.bsky Lexicon)" -LABEL org.opencontainers.image.licenses="MIT OR Apache-2.0" diff --git a/cmd/palomar/Dockerfile.opensearch b/cmd/palomar/Dockerfile.opensearch deleted file mode 100644 index 079d4db2..00000000 --- a/cmd/palomar/Dockerfile.opensearch +++ /dev/null @@ -1,3 +0,0 @@ -FROM opensearchproject/opensearch:2.13.0 -RUN /usr/share/opensearch/bin/opensearch-plugin install --batch analysis-icu -RUN /usr/share/opensearch/bin/opensearch-plugin install --batch analysis-kuromoji diff --git a/cmd/palomar/Dockerfile.opensearch-dashboards b/cmd/palomar/Dockerfile.opensearch-dashboards deleted file mode 100644 index 1ab7970a..00000000 --- a/cmd/palomar/Dockerfile.opensearch-dashboards +++ /dev/null @@ -1,2 +0,0 @@ -FROM opensearchproject/opensearch-dashboards:2.13.0 -RUN /usr/share/opensearch-dashboards/bin/opensearch-dashboards-plugin remove securityDashboards diff --git a/cmd/palomar/README.md b/cmd/palomar/README.md deleted file mode 100644 index d830c2c9..00000000 --- a/cmd/palomar/README.md +++ /dev/null @@ -1,94 +0,0 @@ -# Palomar - -Palomar is a backend search service for atproto, specifically the `bsky.app` post and profile record types. It works by consuming a repo event stream ("firehose") and updating an OpenSearch cluster (fork of Elasticsearch) with docs. - -Almost all the code for this service is actually in the `search/` directory at the top of this repo. - -In September 2023, this service was substantially re-written. It no longer stores records in a local database, returns only "skeleton" results (list of ATURIs or DIDs) via the HTTP API, and defines index mappings. - - -## Query String Syntax - -Currently only a simple query string syntax is supported. Double-quotes can surround phrases, `-` prefix negates a single keyword, and the following initial filters are supported: - -- `from:` will filter to results from that account, based on current (cached) identity resolution -- entire DIDs as an un-quoted keyword will result in filtering to results from that account - - -## Configuration - -Palomar uses environment variables for configuration. - -- `ATP_RELAY_HOST`: URL of firehose to subscribe to, either global Relay or individual PDS (default: `wss://bsky.network`) -- `ATP_PLC_HOST`: PLC directory for identity lookups (default: `https://plc.directory`) -- `DATABASE_URL`: connection string for database to persist firehose cursor subscription state -- `PALOMAR_BIND`: IP/port to have HTTP API listen on (default: `:3999`) -- `ES_USERNAME`: Elasticsearch username (default: `admin`) -- `ES_PASSWORD`: Password for Elasticsearch authentication -- `ES_CERT_FILE`: Optional, for TLS connections -- `ES_HOSTS`: Comma-separated list of Elasticsearch endpoints -- `ES_POST_INDEX`: name of index for post docs (default: `palomar_post`) -- `ES_PROFILE_INDEX`: name of index for profile docs (default: `palomar_profile`) -- `PALOMAR_READONLY`: Set this if the instance should act as a readonly HTTP server (no indexing) - -## HTTP API - -### Query Posts: `/xrpc/app.bsky.unspecced.searchPostsSkeleton` - -HTTP Query Params: - -- `q`: query string, required -- `limit`: integer, default 25 -- `cursor`: string, for partial pagination (uses offset, not a scroll) - -Response: - -- `posts`: array of AT-URI strings -- `hits_total`: integer; optional number of search hits (may not be populated for large result sets, eg over 10k hits) -- `cursor`: string; optionally included if there are more results that can be paginated - -### Query Profiles: `/xrpc/app.bsky.unspecced.searchActorsSkeleton` - -HTTP Query Params: - -- `q`: query string, required -- `limit`: integer, default 25 -- `cursor`: string, for partial pagination (uses offset, not a scroll) -- `typeahead`: boolean, for typeahead behavior (vs. full search) - -Response: - -- `actors`: array of AT-URI strings -- `hits_total`: integer; optional number of search hits (may not be populated for large result sets, eg over 10k hits) -- `cursor`: string; optionally included if there are more results that can be paginated - -## Development Quickstart - -Run an ephemeral opensearch instance on local port 9200, with SSL disabled, and the `analysis-icu` and `analysis-kuromoji` plugins installed, using docker: - - docker build -f Dockerfile.opensearch . -t opensearch-palomar - - # in any non-development system, obviously change this default password - docker run -p 9200:9200 -p 9600:9600 -e "discovery.type=single-node" -e "plugins.security.disabled=true" -e OPENSEARCH_INITIAL_ADMIN_PASSWORD=0penSearch-Pal0mar opensearch-palomar - -See [README.opensearch.md]() for more Opensearch operational tips. - -From the top level of the repository: - - # run combined indexing and search service - make run-dev-search - - # run just the search service - READONLY=true make run-dev-search - -You'll need to get some content in to the index. An easy way to do this is to have palomar consume from the public production firehose. - -You can run test queries from the top level of the repository: - - go run ./cmd/palomar search-post "hello" - go run ./cmd/palomar search-profile "hello" - go run ./cmd/palomar search-profile -typeahead "h" - -For more commands and args: - - go run ./cmd/palomar --help diff --git a/cmd/palomar/README.opensearch.md b/cmd/palomar/README.opensearch.md deleted file mode 100644 index 553c7b31..00000000 --- a/cmd/palomar/README.opensearch.md +++ /dev/null @@ -1,92 +0,0 @@ - -# Basic OpenSearch Operations - -We use OpenSearch version 2.13+, with the `analysis-icu` and `analysis-kuromoji` plugins. These are included automatically on the AWS hosted version of Opensearch, otherwise you need to install: - - sudo /usr/share/opensearch/bin/opensearch-plugin install analysis-icu - sudo /usr/share/opensearch/bin/opensearch-plugin install analysis-kuromoji - sudo service opensearch restart - -If you are trying to use Elasticsearch 7.10 instead of OpenSearch, you can install the plugin with: - - sudo /usr/share/elasticsearch/bin/elasticsearch-plugin install analysis-icu - sudo /usr/share/elasticsearch/bin/elasticsearch-plugin install analysis-kuromoji - sudo service elasticsearch restart - -## Local Development - -With OpenSearch running locally. - -To manually drop and re-build the indices with new schemas (palomar will create these automatically if they don't exist, but this can be helpful when developing the schema itself): - - http delete :9200/palomar_post - http delete :9200/palomar_profile - http put :9200/palomar_post < post_schema.json - http put :9200/palomar_profile < profile_schema.json - -Put a single object (good for debugging): - - head -n1 examples.json | http post :9200/palomar_post/_doc/0 - http get :9200/palomar_post/_doc/0 - -Bulk insert from a file on disk: - - # esbulk is a golang CLI tool which must be installed separately - esbulk -verbose -id ident -index palomar_post -type _doc examples.json - -## Index Aliases - -To make re-indexing and schema changes easier, we can create versioned (or -time-stamped) elasticsearch indexes, and then point to them using index -aliases. The index alias updates are fast and atomic, so we can slowly build up -a new index and then cut over with no downtime. - - http put :9200/palomar_post_v04 < post_schema.json - -To do an atomic swap from one alias to a new one ("zero downtime"): - - http post :9200/_aliases << EOF - { - "actions": [ - { "remove": { "index": "palomar_post_v05", "alias": "palomar_post" }}, - { "add": { "index": "palomar_post_v06", "alias": "palomar_post" }} - ] - } - EOF - -To replace an existing ("real") index with an alias pointer, do two actions -(not truly zero-downtime, but pretty fast): - - http delete :9200/palomar_post - http put :9200/palomar_post_v03/_alias/palomar_post - -## Full-Text Querying - -A generic full-text "query string" query look like this (replace "blood" with -actual query string, and "size" field with the max results to return): - - GET /palomar_post/_search - { - "query": { - "query_string": { - "query": "blood", - "analyzer": "textIcuSearch", - "default_operator": "AND", - "analyze_wildcard": true, - "lenient": true, - "fields": ["handle^5", "text"] - } - }, - "size": 3 - } - -In the results take `.hits.hits[]._source` as the objects; `.hits.total` is the -total number of search hits. - - -## Index Debugging - -Check index size: - - http get :9200/palomar_post/_count - http get :9200/palomar_profile/_count diff --git a/cmd/palomar/docker-compose.yml b/cmd/palomar/docker-compose.yml deleted file mode 100644 index ad8642c4..00000000 --- a/cmd/palomar/docker-compose.yml +++ /dev/null @@ -1,116 +0,0 @@ -version: "3.9" -services: - opensearch: - container_name: opensearch - build: - context: ../../ - dockerfile: cmd/palomar/Dockerfile.opensearch - ports: - - "9200:9200" - - "9600:9600" - environment: - - "discovery.type=single-node" - - "cluster.name=opensearch-palomar" - - "plugins.security.disabled=true" - - "bootstrap.memory_lock=true" # Disable JVM heap memory swapping - - "OPENSEARCH_JAVA_OPTS=-Xms4096m -Xmx4096m" # Set min and max JVM heap sizes to at least 50% of system RAM - - "OPENSEARCH_INITIAL_ADMIN_PASSWORD=0penSearch-Pal0mar" - ulimits: - memlock: - soft: -1 - hard: -1 - nofile: - soft: 65536 - hard: 65536 - volumes: - - type: bind - source: ../../data/opensearch - target: /usr/share/opensearch/data - indexer: - container_name: indexer - build: - context: ../../ - dockerfile: cmd/palomar/Dockerfile - environment: - - "GOLOG_LOG_LEVEL=info" - - "ATP_PLC_HOST=https://plc.directory" - - "ATP_BGS_HOST=wss://bsky.network" - - "ELASTIC_HOSTS=http://opensearch:9200" - - "ES_INSECURE_SSL=true" - - "ENVIRONMENT=dev" - - "ES_POST_INDEX=palomar_post_dev" - - "ES_PROFILE_INDEX=palomar_profile_dev" - - "PALOMAR_DISCOVER_REPOS=false" - - "PALOMAR_BGS_SYNC_RATE_LIMIT=20" - - "PALOMAR_INDEX_MAX_CONCURRENCY=5" - - "DATABASE_URL=sqlite:///data/palomar/search.db" - - "PALOMAR_BIND=:3997" - - "PALOMAR_METRICS_LISTEN=:3996" - depends_on: - - opensearch - ports: - - "3997:3997" - - "3996:3996" - volumes: - - type: bind - source: ../../data - target: /data - # pagerank: - # container_name: pagerank - # build: - # context: ../../ - # dockerfile: cmd/palomar/Dockerfile - # environment: - # - "GOLOG_LOG_LEVEL=info" - # - "ATP_PLC_HOST=https://plc.directory" - # - "ATP_BGS_HOST=wss://bsky.network" - # - "ELASTIC_HOSTS=http://opensearch:9200" - # - "ES_INSECURE_SSL=true" - # - "ENVIRONMENT=dev" - # - "ES_POST_INDEX=palomar_post_dev" - # - "ES_PROFILE_INDEX=palomar_profile_dev" - # - "PALOMAR_DISCOVER_REPOS=false" - # - "PALOMAR_BGS_SYNC_RATE_LIMIT=20" - # - "PALOMAR_INDEX_MAX_CONCURRENCY=5" - # - "DATABASE_URL=sqlite:///data/palomar/pagerank.db" - # - "PAGERANK_FILE=/data/palomar/pageranks.csv" - # depends_on: - # - opensearch - # volumes: - # - type: bind - # source: ../../data - # target: /data - api: - container_name: api - build: - context: ../../ - dockerfile: cmd/palomar/Dockerfile - ports: - - "3999:3999" - - "3998:3998" - environment: - - "GOLOG_LOG_LEVEL=info" - - "ATP_PLC_HOST=https://plc.directory" - - "ATP_BGS_HOST=wss://bsky.network" - - "ELASTIC_HOSTS=http://opensearch:9200" - - "ES_INSECURE_SSL=true" - - "ENVIRONMENT=dev" - - "ES_POST_INDEX=palomar_post_dev" - - "ES_PROFILE_INDEX=palomar_profile_dev" - - "DATABASE_URL=sqlite:///data/palomar/search.db" - - "PALOMAR_READONLY=true" - volumes: - - type: bind - source: ../../data - target: /data - opensearch-dashboards: - build: - context: ../../ - dockerfile: cmd/palomar/Dockerfile.opensearch-dashboards - container_name: opensearch-dashboards - ports: - - 5601:5601 - environment: - OPENSEARCH_HOSTS: '["http://opensearch:9200"]' -networks: - default: diff --git a/cmd/palomar/main.go b/cmd/palomar/main.go deleted file mode 100644 index 556cb3a9..00000000 --- a/cmd/palomar/main.go +++ /dev/null @@ -1,514 +0,0 @@ -package main - -import ( - "context" - "crypto/tls" - "encoding/json" - "fmt" - "log" - "log/slog" - "net/http" - "os" - "strings" - "time" - - _ "github.com/joho/godotenv/autoload" - - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/search" - "github.com/bluesky-social/indigo/util/cliutil" - - "github.com/earthboundkid/versioninfo/v2" - es "github.com/opensearch-project/opensearch-go/v2" - "github.com/urfave/cli/v3" - "go.opentelemetry.io/otel" - "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp" - "go.opentelemetry.io/otel/sdk/resource" - tracesdk "go.opentelemetry.io/otel/sdk/trace" - semconv "go.opentelemetry.io/otel/semconv/v1.4.0" - "golang.org/x/time/rate" -) - -func main() { - if err := run(os.Args); err != nil { - slog.Error("exiting", "err", err) - os.Exit(-1) - } -} - -func run(args []string) error { - - app := cli.Command{ - Name: "palomar", - Usage: "search indexing and query service (using ES or OS)", - Version: versioninfo.Short(), - } - - app.Flags = []cli.Flag{ - &cli.StringFlag{ - Name: "elastic-cert-file", - Usage: "certificate file path", - Sources: cli.EnvVars("ES_CERT_FILE", "ELASTIC_CERT_FILE"), - }, - &cli.BoolFlag{ - Name: "elastic-insecure-ssl", - Usage: "if true, disable SSL cert validation", - Sources: cli.EnvVars("ES_INSECURE_SSL"), - }, - &cli.StringFlag{ - Name: "elastic-username", - Usage: "elasticsearch username", - Value: "admin", - Sources: cli.EnvVars("ES_USERNAME", "ELASTIC_USERNAME"), - }, - &cli.StringFlag{ - Name: "elastic-password", - Usage: "elasticsearch password", - Value: "0penSearch-Pal0mar", - Sources: cli.EnvVars("ES_PASSWORD", "ELASTIC_PASSWORD"), - }, - &cli.StringFlag{ - Name: "elastic-hosts", - Usage: "elasticsearch hosts (schema/host/port)", - Value: "http://localhost:9200", - Sources: cli.EnvVars("ES_HOSTS", "ELASTIC_HOSTS", "OPENSEARCH_URL", "ELASTICSEARCH_URL"), - }, - &cli.StringFlag{ - Name: "es-post-index", - Usage: "ES index for 'post' documents", - Value: "palomar_post", - Sources: cli.EnvVars("ES_POST_INDEX"), - }, - &cli.StringFlag{ - Name: "es-profile-index", - Usage: "ES index for 'profile' documents", - Value: "palomar_profile", - Sources: cli.EnvVars("ES_PROFILE_INDEX"), - }, - &cli.StringFlag{ - Name: "atp-relay-host", - Usage: "hostname and port of Relay to subscribe to", - Value: "wss://bsky.network", - Sources: cli.EnvVars("ATP_RELAY_HOST", "ATP_BGS_HOST"), - }, - &cli.StringFlag{ - Name: "atp-plc-host", - Usage: "method, hostname, and port of PLC registry", - Value: "https://plc.directory", - Sources: cli.EnvVars("ATP_PLC_HOST"), - }, - &cli.IntFlag{ - Name: "max-metadb-connections", - Sources: cli.EnvVars("MAX_METADB_CONNECTIONS"), - Value: 40, - }, - &cli.StringFlag{ - Name: "log-level", - Usage: "log level (debug, info, warn, error)", - Value: "info", - Sources: cli.EnvVars("GOLOG_LOG_LEVEL", "LOG_LEVEL"), - }, - } - - app.Commands = []*cli.Command{ - runCmd, - elasticCheckCmd, - searchPostCmd, - searchProfileCmd, - } - - return app.Run(context.Background(), args) -} - -var runCmd = &cli.Command{ - Name: "run", - Usage: "combined indexing+query server", - Flags: []cli.Flag{ - &cli.StringFlag{ - Name: "database-url", - Value: "sqlite://data/palomar/search.db", - Sources: cli.EnvVars("DATABASE_URL"), - }, - &cli.BoolFlag{ - Name: "readonly", - Sources: cli.EnvVars("PALOMAR_READONLY", "READONLY"), - }, - &cli.StringFlag{ - Name: "bind", - Usage: "IP or address, and port, to listen on for HTTP APIs", - Value: ":3999", - Sources: cli.EnvVars("PALOMAR_BIND"), - }, - &cli.StringFlag{ - Name: "metrics-listen", - Usage: "IP or address, and port, to listen on for metrics APIs", - Value: ":3998", - Sources: cli.EnvVars("PALOMAR_METRICS_LISTEN"), - }, - &cli.IntFlag{ - Name: "relay-sync-rate-limit", - Usage: "max repo sync (checkout) requests per second to upstream (Relay)", - Value: 8, - Sources: cli.EnvVars("PALOMAR_RELAY_SYNC_RATE_LIMIT", "PALOMAR_BGS_SYNC_RATE_LIMIT"), - }, - &cli.IntFlag{ - Name: "index-max-concurrency", - Usage: "max number of concurrent index requests (HTTP POST) to search index", - Value: 20, - Sources: cli.EnvVars("PALOMAR_INDEX_MAX_CONCURRENCY"), - }, - &cli.IntFlag{ - Name: "indexing-rate-limit", - Usage: "max number of documents per second to index", - Value: 50_000, - Sources: cli.EnvVars("PALOMAR_INDEXING_RATE_LIMIT"), - }, - &cli.IntFlag{ - Name: "plc-rate-limit", - Usage: "max number of requests per second to PLC registry", - Value: 100, - Sources: cli.EnvVars("PALOMAR_PLC_RATE_LIMIT"), - }, - &cli.BoolFlag{ - Name: "discover-repos", - Usage: "if true, discover repositories from the Relay", - Sources: cli.EnvVars("PALOMAR_DISCOVER_REPOS"), - Value: false, - }, - &cli.StringFlag{ - Name: "pagerank-file", - Sources: cli.EnvVars("PAGERANK_FILE"), - }, - &cli.StringFlag{ - Name: "bulk-posts-file", - Sources: cli.EnvVars("BULK_POSTS_FILE"), - }, - &cli.StringFlag{ - Name: "bulk-profiles-file", - Sources: cli.EnvVars("BULK_PROFILES_FILE"), - }, - }, - Action: func(ctx context.Context, cmd *cli.Command) error { - logLevel := slog.LevelInfo - switch cmd.String("log-level") { - case "debug": - logLevel = slog.LevelDebug - case "info": - logLevel = slog.LevelInfo - case "warn": - logLevel = slog.LevelWarn - case "error": - logLevel = slog.LevelError - } - - logger := slog.New(slog.NewJSONHandler(os.Stdout, &slog.HandlerOptions{ - Level: logLevel, - AddSource: true, - })) - slog.SetDefault(logger) - - readonly := cmd.Bool("readonly") - - // Enable OTLP HTTP exporter - // For relevant environment variables: - // https://pkg.go.dev/go.opentelemetry.io/otel/exporters/otlp/otlptrace#readme-environment-variables - // At a minimum, you need to set - // OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4318 - if ep := os.Getenv("OTEL_EXPORTER_OTLP_ENDPOINT"); ep != "" { - slog.Info("setting up trace exporter", "endpoint", ep) - ctx, cancel := context.WithCancel(context.Background()) - defer cancel() - - exp, err := otlptracehttp.New(ctx) - if err != nil { - log.Fatal("failed to create trace exporter", "error", err) - } - defer func() { - ctx, cancel := context.WithTimeout(context.Background(), time.Second) - defer cancel() - if err := exp.Shutdown(ctx); err != nil { - slog.Error("failed to shutdown trace exporter", "error", err) - } - }() - - tp := tracesdk.NewTracerProvider( - tracesdk.WithBatcher(exp), - tracesdk.WithResource(resource.NewWithAttributes( - semconv.SchemaURL, - semconv.ServiceNameKey.String("palomar"), - attribute.String("env", os.Getenv("ENVIRONMENT")), // DataDog - attribute.String("environment", os.Getenv("ENVIRONMENT")), // Others - attribute.Int64("ID", 1), - )), - ) - otel.SetTracerProvider(tp) - } - - escli, err := createEsClient(cmd) - if err != nil { - return fmt.Errorf("failed to get elasticsearch: %w", err) - } - - base := identity.BaseDirectory{ - PLCURL: cmd.String("atp-plc-host"), - HTTPClient: http.Client{ - Timeout: time.Second * 15, - }, - PLCLimiter: rate.NewLimiter(rate.Limit(cmd.Int("plc-rate-limit")), 1), - TryAuthoritativeDNS: true, - SkipDNSDomainSuffixes: []string{".bsky.social"}, - } - dir := identity.NewCacheDirectory(&base, 1_500_000, time.Hour*24, time.Minute*2, time.Minute*5) - - apiConfig := search.ServerConfig{ - Logger: logger, - ProfileIndex: cmd.String("es-profile-index"), - PostIndex: cmd.String("es-post-index"), - } - - srv, err := search.NewServer(escli, dir, apiConfig) - if err != nil { - return err - } - - // Configure the indexer if we're not in readonly mode - if !readonly { - db, err := cliutil.SetupDatabase(cmd.String("database-url"), cmd.Int("max-metadb-connections")) - if err != nil { - return fmt.Errorf("failed to set up database: %w", err) - } - - indexerConfig := search.IndexerConfig{ - RelayHost: cmd.String("atp-relay-host"), - ProfileIndex: cmd.String("es-profile-index"), - PostIndex: cmd.String("es-post-index"), - Logger: logger, - RelaySyncRateLimit: cmd.Int("relay-sync-rate-limit"), - IndexMaxConcurrency: cmd.Int("index-max-concurrency"), - DiscoverRepos: cmd.Bool("discover-repos"), - IndexingRateLimit: cmd.Int("indexing-rate-limit"), - } - - idx, err := search.NewIndexer(db, escli, dir, indexerConfig) - if err != nil { - return fmt.Errorf("failed to set up indexer: %w", err) - } - - srv.Indexer = idx - } - - go func() { - if err := srv.RunMetrics(cmd.String("metrics-listen")); err != nil { - slog.Error("failed to start metrics endpoint", "error", err) - panic(fmt.Errorf("failed to start metrics endpoint: %w", err)) - } - }() - - go func() { - srv.RunAPI(cmd.String("bind")) - }() - - // If we're in readonly mode, just block forever - if readonly { - select {} - } else if cmd.String("pagerank-file") != "" && srv.Indexer != nil { - // If we're not in readonly mode, and we have a pagerank file, update pageranks - ctx := context.Background() - if err := srv.Indexer.BulkIndexPageranks(ctx, cmd.String("pagerank-file")); err != nil { - return fmt.Errorf("failed to update pageranks: %w", err) - } - } else if cmd.String("bulk-posts-file") != "" && srv.Indexer != nil { - // If we're not in readonly mode, and we have a bulk posts file, index posts - ctx := context.Background() - if err := srv.Indexer.BulkIndexPosts(ctx, cmd.String("bulk-posts-file")); err != nil { - return fmt.Errorf("failed to bulk index posts: %w", err) - } - } else if cmd.String("bulk-profiles-file") != "" && srv.Indexer != nil { - // If we're not in readonly mode, and we have a bulk profiles file, index profiles - ctx := context.Background() - if err := srv.Indexer.BulkIndexProfiles(ctx, cmd.String("bulk-profiles-file")); err != nil { - return fmt.Errorf("failed to bulk index profiles: %w", err) - } - } else if srv.Indexer != nil { - // Otherwise, just run the indexer - ctx := context.Background() - if err := srv.Indexer.EnsureIndices(ctx); err != nil { - return fmt.Errorf("failed to create opensearch indices: %w", err) - } - if err := srv.Indexer.RunIndexer(ctx); err != nil { - return fmt.Errorf("failed to run indexer: %w", err) - } - } - - return nil - }, -} - -var elasticCheckCmd = &cli.Command{ - Name: "elastic-check", - Flags: []cli.Flag{}, - Action: func(ctx context.Context, cmd *cli.Command) error { - escli, err := createEsClient(cmd) - if err != nil { - return err - } - - // NOTE: this extra info check is redundant; createEsClient() already made this call and logged results - inf, err := escli.Info() - if err != nil { - return fmt.Errorf("failed to get info: %w", err) - } - defer inf.Body.Close() - if inf.IsError() { - return fmt.Errorf("failed to get info") - } - slog.Info("opensearch client connected", "client_info", inf) - - resp, err := escli.Indices.Exists([]string{cmd.String("es-profile-index"), cmd.String("es-post-index")}) - if err != nil { - return fmt.Errorf("failed to check index existence: %w", err) - } - defer resp.Body.Close() - if inf.IsError() { - return fmt.Errorf("failed to check index existence") - } - slog.Info("index existence", "resp", resp) - - return nil - - }, -} - -func printHits(resp *search.EsSearchResponse) { - fmt.Printf("%d hits in %d\n", len(resp.Hits.Hits), resp.Took) - for _, hit := range resp.Hits.Hits { - b, _ := json.Marshal(hit.Source) - fmt.Println(string(b)) - } - return -} - -var searchPostCmd = &cli.Command{ - Name: "search-post", - Usage: "run a simple query against posts index", - Action: func(ctx context.Context, cmd *cli.Command) error { - escli, err := createEsClient(cmd) - if err != nil { - return err - } - res, err := search.DoSearchPosts( - context.Background(), - identity.DefaultDirectory(), // TODO: parse PLC arg - escli, - cmd.String("es-post-index"), - &search.PostSearchParams{ - Query: strings.Join(cmd.Args().Slice(), " "), - Offset: 0, - Size: 20, - }, - ) - if err != nil { - return err - } - printHits(res) - return nil - }, -} - -var searchProfileCmd = &cli.Command{ - Name: "search-profile", - Usage: "run a simple query against posts index", - Flags: []cli.Flag{ - &cli.BoolFlag{ - Name: "typeahead", - }, - }, - Action: func(ctx context.Context, cmd *cli.Command) error { - escli, err := createEsClient(cmd) - if err != nil { - return err - } - if cmd.Bool("typeahead") { - res, err := search.DoSearchProfilesTypeahead( - context.Background(), - escli, - cmd.String("es-profile-index"), - &search.ActorSearchParams{ - Query: strings.Join(cmd.Args().Slice(), " "), - Size: 10, - }, - ) - if err != nil { - return err - } - printHits(res) - } else { - res, err := search.DoSearchProfiles( - context.Background(), - identity.DefaultDirectory(), // TODO: parse PLC arg - escli, - cmd.String("es-profile-index"), - &search.ActorSearchParams{ - Query: strings.Join(cmd.Args().Slice(), " "), - Offset: 0, - Size: 20, - }, - ) - if err != nil { - return err - } - printHits(res) - } - return nil - }, -} - -func createEsClient(cmd *cli.Command) (*es.Client, error) { - - addrs := []string{} - if hosts := cmd.String("elastic-hosts"); hosts != "" { - addrs = strings.Split(hosts, ",") - } - - certfi := cmd.String("elastic-cert-file") - var cert []byte - if certfi != "" { - b, err := os.ReadFile(certfi) - if err != nil { - return nil, err - } - - cert = b - } - - insecure := cmd.Bool("elastic-insecure-ssl") - - cfg := es.Config{ - Addresses: addrs, - Username: cmd.String("elastic-username"), - Password: cmd.String("elastic-password"), - CACert: cert, - Transport: &http.Transport{ - Proxy: http.ProxyFromEnvironment, - MaxIdleConnsPerHost: 20, - TLSClientConfig: &tls.Config{ - InsecureSkipVerify: insecure, - }, - }, - } - - escli, err := es.NewClient(cfg) - if err != nil { - return nil, fmt.Errorf("failed to set up client: %w", err) - } - - info, err := escli.Info() - if err != nil { - return nil, fmt.Errorf("cannot get escli info: %w", err) - } - defer info.Body.Close() - slog.Debug("opensearch client initialized", "info", info) - - return escli, nil -} diff --git a/cmd/palomar/opensearch_dashboards.yml b/cmd/palomar/opensearch_dashboards.yml deleted file mode 100644 index aa93bedf..00000000 --- a/cmd/palomar/opensearch_dashboards.yml +++ /dev/null @@ -1,4 +0,0 @@ ---- -server.name: opensearch-dashboards -server.host: "0.0.0.0" -opensearch.hosts: http://opensearch:9200 diff --git a/cmd/palomar/pagerank.sh b/cmd/palomar/pagerank.sh deleted file mode 100644 index 255332a7..00000000 --- a/cmd/palomar/pagerank.sh +++ /dev/null @@ -1,57 +0,0 @@ -#!/bin/bash -set -o errexit -set -o nounset -set -o pipefail - -export SCYLLA_KEYSPACE="${SCYLLA_KEYSPACE:-}" -export SCYLLA_HOST="${SCYLLA_HOST:-}" - -# Used by pagerank. -export FOLLOWS_FILE="/data/follows.csv" -export ACTORS_FILE="/data/actors.csv" -export OUTPUT_FILE="/data/pageranks.csv" -export EXPECTED_ACTOR_COUNT="5000000" -export RUST_LOG="info" - -# Used by palomar. -export PAGERANK_FILE="${OUTPUT_FILE}" -export PALOMAR_INDEXING_RATE_LIMIT="10000" - -function run_pagerank { - # Check that the required environment variables are set. - if [[ "${SCYLLA_KEYSPACE}" == "" ]]; then - echo "SCYLLA_KEYSPACE is not set" - exit 1 - fi - - if [[ "${SCYLLA_HOST}" == "" ]]; then - echo "SCYLLA_HOST is not set" - exit 1 - fi - - # Dump the tables to CSV files. - rm --force "${FOLLOWS_FILE}" - cqlsh \ - "--keyspace=${SCYLLA_KEYSPACE}" \ - "--request-timeout=1200" \ - ---execute "COPY follows (actor_did, subject_did) TO '${FOLLOWS_FILE}' WITH HEADER = FALSE;" \ - "${SCYLLA_HOST}" - - rm --force "${ACTORS_FILE}" - cqlsh \ - "--keyspace=${SCYLLA_KEYSPACE}" \ - "--request-timeout=1200" \ - ---execute "COPY actors (did) TO '${ACTORS_FILE}' WITH HEADER = FALSE;" \ - "${SCYLLA_HOST}" - - # Run the pagerank file which reads in the table CSV files and outputs a CSV. - /usr/local/bin/pagerank - - # Run palomar with the pagerank CSV file. - /palomar run -} - -while true; do - run_pagerank - sleep 24h -done diff --git a/search/bulk.go b/search/bulk.go deleted file mode 100644 index bb58e0b6..00000000 --- a/search/bulk.go +++ /dev/null @@ -1,365 +0,0 @@ -package search - -import ( - "bufio" - "context" - "encoding/hex" - "encoding/json" - "fmt" - "os" - "strconv" - "strings" - "sync" - - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - - "github.com/ipfs/go-cid" -) - -type pagerankJob struct { - did syntax.DID - rank float64 -} - -// BulkIndexPageranks updates the pageranks for the DIDs in the Search Index from a CSV file. -func (idx *Indexer) BulkIndexPageranks(ctx context.Context, pagerankFile string) error { - f, err := os.Open(pagerankFile) - if err != nil { - return fmt.Errorf("failed to open csv file: %w", err) - } - defer f.Close() - - // Run 5 pagerank indexers in parallel - for range 5 { - go idx.runPagerankIndexer(ctx) - } - - logger := idx.logger.With("source", "bulk_index_pageranks") - - queue := make(chan string, 20_000) - wg := &sync.WaitGroup{} - workerCount := 20 - for range workerCount { - wg.Go(func() { - for line := range queue { - if err := idx.processPagerankCSVLine(line); err != nil { - logger.Error("failed to process line", "err", err) - } - } - }) - } - - // Create a scanner to read the file line by line - scanner := bufio.NewScanner(f) - buf := make([]byte, 0, 64*1024) - scanner.Buffer(buf, 1024*1024) - - linesRead := 0 - - // Iterate over each line in the file - for scanner.Scan() { - line := scanner.Text() - - queue <- line - - linesRead++ - if linesRead%100_000 == 0 { - idx.logger.Info("processed csv lines", "lines", linesRead) - } - } - - close(queue) - - // Check for any scanner errors - if err := scanner.Err(); err != nil { - return fmt.Errorf("error reading csv file: %w", err) - } - - wg.Wait() - - idx.logger.Info("finished processing csv file", "lines", linesRead) - - return nil -} - -// BulkIndexPosts indexes posts from a CSV file. -func (idx *Indexer) BulkIndexPosts(ctx context.Context, postsFile string) error { - f, err := os.Open(postsFile) - if err != nil { - return fmt.Errorf("failed to open csv file: %w", err) - } - defer f.Close() - - // Run 5 post indexers in parallel - for range 5 { - go idx.runPostIndexer(ctx) - } - - logger := idx.logger.With("source", "bulk_index_posts") - - queue := make(chan string, 20_000) - wg := &sync.WaitGroup{} - workerCount := 20 - for range workerCount { - wg.Go(func() { - for line := range queue { - if err := idx.processPostCSVLine(line); err != nil { - logger.Error("failed to process line", "err", err) - } - } - }) - } - - // Create a scanner to read the file line by line - scanner := bufio.NewScanner(f) - buf := make([]byte, 0, 64*1024) - scanner.Buffer(buf, 1024*1024) - - linesRead := 0 - - // Iterate over each line in the file - for scanner.Scan() { - line := scanner.Text() - - queue <- line - - linesRead++ - if linesRead%100_000 == 0 { - idx.logger.Info("processed csv lines", "lines", linesRead) - } - } - - close(queue) - - // Check for any scanner errors - if err := scanner.Err(); err != nil { - return fmt.Errorf("error reading csv file: %w", err) - } - - wg.Wait() - - idx.logger.Info("finished processing csv file", "lines", linesRead) - - return nil -} - -// BulkIndexProfiles indexes profiles from a CSV file. -func (idx *Indexer) BulkIndexProfiles(ctx context.Context, profilesFile string) error { - f, err := os.Open(profilesFile) - if err != nil { - return fmt.Errorf("failed to open csv file: %w", err) - } - defer f.Close() - - for range 5 { - go idx.runProfileIndexer(ctx) - } - - logger := idx.logger.With("source", "bulk_index_profiles") - - queue := make(chan string, 20_000) - wg := &sync.WaitGroup{} - workerCount := 20 - for range workerCount { - wg.Go(func() { - for line := range queue { - if err := idx.processProfileCSVLine(line); err != nil { - logger.Error("failed to process line", "err", err) - } - } - }) - } - - // Create a scanner to read the file line by line - scanner := bufio.NewScanner(f) - buf := make([]byte, 0, 64*1024) - scanner.Buffer(buf, 1024*1024) - - linesRead := 0 - - // Iterate over each line in the file - for scanner.Scan() { - line := scanner.Text() - - queue <- line - - linesRead++ - if linesRead%100_000 == 0 { - idx.logger.Info("processed csv lines", "lines", linesRead) - } - } - - close(queue) - - // Check for any scanner errors - if err := scanner.Err(); err != nil { - return fmt.Errorf("error reading csv file: %w", err) - } - - wg.Wait() - - idx.logger.Info("finished processing csv file", "lines", linesRead) - - return nil -} - -func (idx *Indexer) processPagerankCSVLine(line string) error { - // Split the line into DID and rank - parts := strings.Split(line, ",") - if len(parts) != 2 { - return fmt.Errorf("invalid pagerank line: %s", line) - } - - did, err := syntax.ParseDID(parts[0]) - if err != nil { - return fmt.Errorf("invalid DID: %s", parts[0]) - } - - rank, err := strconv.ParseFloat(parts[1], 64) - if err != nil { - return fmt.Errorf("invalid pagerank value: %s", parts[1]) - } - - job := PagerankIndexJob{ - did: did, - rank: rank, - } - - // Send the job to the pagerank queue - idx.pagerankQueue <- &job - - return nil -} - -func (idx *Indexer) processPostCSVLine(line string) error { - // CSV is formatted as - // actor_did,rkey,taken_down(time or null),violates_threadgate(False or null),cid,raw(post JSON as hex) - parts := strings.Split(line, ",") - if len(parts) != 6 { - return fmt.Errorf("invalid csv line: %s", line) - } - - did, err := syntax.ParseDID(parts[0]) - if err != nil { - return fmt.Errorf("invalid DID: %s", parts[0]) - } - - rkey, err := syntax.ParseRecordKey(parts[1]) - if err != nil { - return fmt.Errorf("invalid record key: %s", parts[1]) - } - - isTakenDown := false - if parts[2] != "" && parts[2] != "null" { - isTakenDown = true - } - - violatesThreadgate := false - if parts[3] != "" && parts[3] != "False" { - violatesThreadgate = true - } - - if isTakenDown || violatesThreadgate { - return nil - } - - cid, err := cid.Parse(parts[4]) - if err != nil { - return fmt.Errorf("invalid CID: %s", parts[4]) - } - - if len(parts[5]) <= 2 { - return nil - } - - raw, err := hex.DecodeString(parts[5][2:]) - if err != nil { - return fmt.Errorf("invalid raw record (%s/%s): %s", did, rkey, parts[5][2:]) - } - - post := appbsky.FeedPost{} - if err := json.Unmarshal(raw, &post); err != nil { - return fmt.Errorf("failed to unmarshal post: %w", err) - } - - job := PostIndexJob{ - did: did, - rkey: rkey.String(), - rcid: cid, - record: &post, - } - - // Send the job to the post queue - idx.postQueue <- &job - - return nil -} - -func (idx *Indexer) processProfileCSVLine(line string) error { - // CSV is formatted as - // actor_did,taken_down(time or null),cid,handle,raw(profile JSON as hex) - parts := strings.Split(line, ",") - if len(parts) != 5 { - return fmt.Errorf("invalid csv line: %s", line) - } - - did, err := syntax.ParseDID(parts[0]) - if err != nil { - return fmt.Errorf("invalid DID: %s", parts[0]) - } - - isTakenDown := false - if parts[1] != "" && parts[1] != "null" { - isTakenDown = true - } - - if isTakenDown { - return nil - } - - // Skip actors without profile records - if parts[2] == "" { - return nil - } - - cid, err := cid.Parse(parts[2]) - if err != nil { - return fmt.Errorf("invalid CID: %s", parts[2]) - } - - if len(parts[3]) <= 2 { - return nil - } - - raw, err := hex.DecodeString(parts[4][2:]) - if err != nil { - return fmt.Errorf("invalid raw record (%s): %s", did, parts[4][2:]) - } - - profile := appbsky.ActorProfile{} - if err := json.Unmarshal(raw, &profile); err != nil { - return fmt.Errorf("failed to unmarshal profile: %w", err) - } - - ident := identity.Identity{DID: did} - - handle, err := syntax.ParseHandle(parts[3]) - if err != nil { - ident.Handle = syntax.HandleInvalid - } else { - ident.Handle = handle - } - - job := ProfileIndexJob{ - ident: &ident, - rcid: cid, - record: &profile, - } - - // Send the job to the profile queue - idx.profileQueue <- &job - - return nil -} diff --git a/search/firehose.go b/search/firehose.go deleted file mode 100644 index 3a35e91c..00000000 --- a/search/firehose.go +++ /dev/null @@ -1,368 +0,0 @@ -package search - -import ( - "bytes" - "context" - "fmt" - "net/http" - "net/url" - "strings" - "time" - - comatproto "github.com/bluesky-social/indigo/api/atproto" - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/syntax" - "github.com/bluesky-social/indigo/backfill" - "github.com/bluesky-social/indigo/events" - "github.com/bluesky-social/indigo/events/schedulers/autoscaling" - lexutil "github.com/bluesky-social/indigo/lex/util" - "github.com/bluesky-social/indigo/repo" - - "github.com/earthboundkid/versioninfo/v2" - "github.com/gorilla/websocket" - "github.com/ipfs/go-cid" - typegen "github.com/whyrusleeping/cbor-gen" -) - -func (idx *Indexer) getLastCursor() (int64, error) { - var lastSeq LastSeq - if err := idx.db.Find(&lastSeq).Error; err != nil { - return 0, err - } - - if lastSeq.ID == 0 { - return 0, idx.db.Create(&lastSeq).Error - } - - return lastSeq.Seq, nil -} - -func (idx *Indexer) updateLastCursor(curs int64) error { - return idx.db.Model(LastSeq{}).Where("id = 1").Update("seq", curs).Error -} - -func (idx *Indexer) RunIndexer(ctx context.Context) error { - cur, err := idx.getLastCursor() - if err != nil { - return fmt.Errorf("get last cursor: %w", err) - } - - // Start the indexer batch workers - go idx.runPostIndexer(ctx) - go idx.runProfileIndexer(ctx) - - err = idx.bfs.LoadJobs(ctx) - if err != nil { - return fmt.Errorf("loading backfill jobs: %w", err) - } - go idx.bf.Start() - - if idx.enableRepoDiscovery { - go idx.discoverRepos() - } - - d := websocket.DefaultDialer - u, err := url.Parse(idx.relayhost) - if err != nil { - return fmt.Errorf("invalid bgshost URI: %w", err) - } - u.Path = "xrpc/com.atproto.sync.subscribeRepos" - if cur != 0 { - u.RawQuery = fmt.Sprintf("cursor=%d", cur) - } - con, _, err := d.Dial(u.String(), http.Header{ - "User-Agent": []string{fmt.Sprintf("palomar/%s", versioninfo.Short())}, - }) - if err != nil { - return fmt.Errorf("events dial failed: %w", err) - } - - rsc := &events.RepoStreamCallbacks{ - RepoCommit: func(evt *comatproto.SyncSubscribeRepos_Commit) error { - ctx := context.Background() - ctx, span := tracer.Start(ctx, "RepoCommit") - defer span.End() - - defer func() { - if evt.Seq%50 == 0 { - if err := idx.updateLastCursor(evt.Seq); err != nil { - idx.logger.Error("failed to persist cursor", "err", err) - } - } - }() - logEvt := idx.logger.With("repo", evt.Repo, "rev", evt.Rev, "seq", evt.Seq) - if evt.TooBig && evt.Since != nil { - // TODO: handle this case (instead of return nil) - logEvt.Error("skipping non-genesis tooBig events for now") - return nil - } - - if evt.TooBig { - if err := idx.processTooBigCommit(ctx, evt); err != nil { - // TODO: handle this case (instead of return nil) - logEvt.Error("failed to process tooBig event", "err", err) - return nil - } - - return nil - } - - // Pass events to the backfiller which will process or buffer as needed - if err := idx.bf.HandleEvent(ctx, evt); err != nil { - logEvt.Error("failed to handle event", "err", err) - } - - return nil - - }, - // TODO: process RepoIdentity - RepoIdentity: func(evt *comatproto.SyncSubscribeRepos_Identity) error { - ctx := context.Background() - ctx, span := tracer.Start(ctx, "RepoIdentity") - defer span.End() - - did, err := syntax.ParseDID(evt.Did) - if err != nil { - idx.logger.Error("bad DID in RepoIdentity event", "did", evt.Did, "seq", evt.Seq, "err", err) - return nil - } - ident, err := idx.dir.LookupDID(ctx, did) - if err != nil { - idx.logger.Error("failed identity resolution in RepoIdentity event", "did", evt.Did, "seq", evt.Seq, "err", err) - return nil - } - if err := idx.updateUserHandle(ctx, did, ident.Handle.String()); err != nil { - // TODO: handle this case (instead of return nil) - idx.logger.Error("failed to update user handle", "did", evt.Did, "handle", ident.Handle, "seq", evt.Seq, "err", err) - } - return nil - }, - } - - return events.HandleRepoStream( - ctx, con, autoscaling.NewScheduler( - autoscaling.DefaultAutoscaleSettings(), - idx.relayhost, - rsc.EventHandler, - ), - idx.logger, - ) -} - -func (idx *Indexer) discoverRepos() { - ctx := context.Background() - log := idx.logger.With("func", "discoverRepos") - log.Info("starting repo discovery") - - cursor := "" - limit := int64(500) - - total := 0 - totalErrored := 0 - - for { - resp, err := comatproto.SyncListRepos(ctx, idx.relayXRPC, cursor, limit) - if err != nil { - log.Error("failed to list repos", "err", err) - time.Sleep(5 * time.Second) - continue - } - log.Info("got repo page", "count", len(resp.Repos), "cursor", resp.Cursor) - errored := 0 - for _, repo := range resp.Repos { - _, err := idx.bfs.GetOrCreateJob(ctx, repo.Did, backfill.StateEnqueued) - if err != nil { - log.Error("failed to get or create job", "did", repo.Did, "err", err) - errored++ - } - } - log.Info("enqueued repos", "total", len(resp.Repos), "errored", errored) - totalErrored += errored - total += len(resp.Repos) - if resp.Cursor != nil && *resp.Cursor != "" { - cursor = *resp.Cursor - } else { - break - } - } - - log.Info("finished repo discovery", "totalJobs", total, "totalErrored", totalErrored) -} - -func (idx *Indexer) handleCreateOrUpdate(ctx context.Context, rawDID string, rev string, path string, recB *[]byte, rcid *cid.Cid) error { - logger := idx.logger.With("func", "handleCreateOrUpdate", "did", rawDID, "rev", rev, "path", path) - // Since this gets called in a backfill job, we need to check if the path is a post or profile - if !strings.Contains(path, "app.bsky.feed.post") && !strings.Contains(path, "app.bsky.actor.profile") { - return nil - } - - did, err := syntax.ParseDID(rawDID) - if err != nil { - return fmt.Errorf("bad DID syntax in event: %w", err) - } - - // CBOR Unmarshal the record - recCBOR, err := lexutil.CborDecodeValue(*recB) - if err != nil { - return fmt.Errorf("cbor decode: %w", err) - } - - rec, ok := recCBOR.(typegen.CBORMarshaler) - if !ok { - return fmt.Errorf("failed to cast record to CBORMarshaler") - } - - parts := strings.SplitN(path, "/", 3) - if len(parts) < 2 { - logger.Warn("skipping post record with malformed path") - return nil - } - - switch rec := rec.(type) { - case *appbsky.FeedPost: - rkey, err := syntax.ParseTID(parts[1]) - if err != nil { - logger.Warn("skipping post record with non-TID rkey") - return nil - } - - job := PostIndexJob{ - did: did, - record: rec, - rcid: *rcid, - rkey: rkey.String(), - } - - // Send the job to the bulk indexer - idx.postQueue <- &job - postsIndexed.Inc() - case *appbsky.ActorProfile: - if parts[1] != "self" { - return nil - } - - ident, err := idx.dir.LookupDID(ctx, did) - if err != nil { - return fmt.Errorf("resolving identity: %w", err) - } - if ident == nil { - return fmt.Errorf("identity not found for did: %s", did.String()) - } - - job := ProfileIndexJob{ - ident: ident, - record: rec, - rcid: *rcid, - } - - // Send the job to the bulk indexer - idx.profileQueue <- &job - profilesIndexed.Inc() - default: - } - return nil -} - -func (idx *Indexer) handleDelete(ctx context.Context, rawDID, rev, path string) error { - // Since this gets called in a backfill job, we need to check if the path is a post or profile - if !strings.Contains(path, "app.bsky.feed.post") && !strings.Contains(path, "app.bsky.actor.profile") { - return nil - } - - did, err := syntax.ParseDID(rawDID) - if err != nil { - return fmt.Errorf("invalid DID in event: %w", err) - } - - switch { - // TODO: handle profile deletes, its an edge case, but worth doing still - case strings.Contains(path, "app.bsky.feed.post"): - if err := idx.deletePost(ctx, did, path); err != nil { - return err - } - postsDeleted.Inc() - case strings.Contains(path, "app.bsky.actor.profile"): - // profilesDeleted.Inc() - } - - return nil -} - -func (idx *Indexer) processTooBigCommit(ctx context.Context, evt *comatproto.SyncSubscribeRepos_Commit) error { - logger := idx.logger.With("func", "processTooBigCommit", "repo", evt.Repo, "rev", evt.Rev, "seq", evt.Seq) - - repodata, err := comatproto.SyncGetRepo(ctx, idx.relayXRPC, evt.Repo, "") - if err != nil { - return err - } - - r, err := repo.ReadRepoFromCar(ctx, bytes.NewReader(repodata)) - if err != nil { - return err - } - - did, err := syntax.ParseDID(evt.Repo) - if err != nil { - return fmt.Errorf("bad DID in repo event: %w", err) - } - - ident, err := idx.dir.LookupDID(ctx, did) - if err != nil { - return err - } - if ident == nil { - return fmt.Errorf("identity not found for did: %s", did.String()) - } - - return r.ForEach(ctx, "", func(k string, v cid.Cid) error { - if strings.HasPrefix(k, "app.bsky.feed.post") || strings.HasPrefix(k, "app.bsky.actor.profile") { - rcid, rec, err := r.GetRecord(ctx, k) - if err != nil { - // TODO: handle this case (instead of return nil) - idx.logger.Error("failed to get record from repo checkout", "path", k, "err", err) - return nil - } - - parts := strings.SplitN(k, "/", 3) - if len(parts) < 2 { - logger.Warn("skipping post record with malformed path") - return nil - } - - switch rec := rec.(type) { - case *appbsky.FeedPost: - rkey, err := syntax.ParseTID(parts[1]) - if err != nil { - logger.Warn("skipping post record with non-TID rkey") - return nil - } - - job := PostIndexJob{ - did: did, - record: rec, - rcid: rcid, - rkey: rkey.String(), - } - - // Send the job to the bulk indexer - idx.postQueue <- &job - case *appbsky.ActorProfile: - if parts[1] != "self" { - return nil - } - - job := ProfileIndexJob{ - ident: ident, - record: rec, - rcid: rcid, - } - - // Send the job to the bulk indexer - idx.profileQueue <- &job - default: - } - - } - return nil - }) -} diff --git a/search/handlers.go b/search/handlers.go deleted file mode 100644 index 9f4e0495..00000000 --- a/search/handlers.go +++ /dev/null @@ -1,473 +0,0 @@ -package search - -import ( - "context" - "encoding/json" - "fmt" - "slices" - "strconv" - "strings" - "sync" - - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/syntax" - - "github.com/labstack/echo/v4" - otel "go.opentelemetry.io/otel" - "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/codes" -) - -var tracer = otel.Tracer("search") - -func parseCursorLimit(e echo.Context) (int, int, error) { - offset := 0 - if c := strings.TrimSpace(e.QueryParam("cursor")); c != "" { - v, err := strconv.Atoi(c) - if err != nil { - return 0, 0, &echo.HTTPError{ - Code: 400, - Message: fmt.Sprintf("invalid value for 'cursor': %s", err), - } - } - offset = v - } - - if offset < 0 { - offset = 0 - } - if offset > 10000 { - return 0, 0, &echo.HTTPError{ - Code: 400, - Message: "invalid value for 'cursor' (can't paginate so deep)", - } - } - - limit := 25 - if l := strings.TrimSpace(e.QueryParam("limit")); l != "" { - v, err := strconv.Atoi(l) - if err != nil { - return 0, 0, &echo.HTTPError{ - Code: 400, - Message: fmt.Sprintf("invalid value for 'count': %s", err), - } - } - - limit = v - } - - if limit > 100 { - limit = 100 - } - if limit < 0 { - limit = 0 - } - return offset, limit, nil -} - -func (s *Server) handleSearchPostsSkeleton(e echo.Context) error { - ctx, span := tracer.Start(e.Request().Context(), "handleSearchPostsSkeleton") - defer span.End() - - span.SetAttributes(attribute.String("query", e.QueryParam("q"))) - - q := strings.TrimSpace(e.QueryParam("q")) - if q == "" { - return e.JSON(400, map[string]any{ - "error": "must pass non-empty search query", - }) - } - - params := PostSearchParams{ - Query: q, - // TODO: parse/validate the sort options here? - Sort: e.QueryParam("sort"), - Domain: e.QueryParam("domain"), - URL: e.QueryParam("url"), - } - - viewerStr := e.QueryParam("viewer") - if viewerStr != "" { - d, err := syntax.ParseDID(viewerStr) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid DID for 'viewer': %s", err), - }) - } - params.Viewer = &d - } - authorStr := e.QueryParam("author") - if authorStr != "" { - atid, err := syntax.ParseAtIdentifier(authorStr) - if err != nil { - return &echo.HTTPError{ - Code: 400, - Message: fmt.Sprintf("invalid DID for 'author': %s", err), - } - } - if atid.IsHandle() { - ident, err := s.dir.Lookup(e.Request().Context(), atid) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid Handle for 'author': %s", err), - }) - } - params.Author = &ident.DID - } else { - d, err := atid.AsDID() - if err != nil { - return err - } - params.Author = &d - } - } - - mentionsStr := e.QueryParam("mentions") - if mentionsStr != "" { - atid, err := syntax.ParseAtIdentifier(mentionsStr) - if err != nil { - return &echo.HTTPError{ - Code: 400, - Message: fmt.Sprintf("invalid DID for 'mentions': %s", err), - } - } - if atid.IsHandle() { - ident, err := s.dir.Lookup(e.Request().Context(), atid) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid Handle for 'mentions': %s", err), - }) - } - params.Mentions = &ident.DID - } else { - d, err := atid.AsDID() - if err != nil { - return err - } - params.Mentions = &d - } - } - - sinceStr := e.QueryParam("since") - if sinceStr != "" { - dt, err := syntax.ParseDatetime(sinceStr) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid Datetime for 'since': %s", err), - }) - } - params.Since = &dt - } - - untilStr := e.QueryParam("until") - if untilStr != "" { - dt, err := syntax.ParseDatetime(untilStr) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid Datetime for 'until': %s", err), - }) - } - params.Until = &dt - } - - langStr := e.QueryParam("lang") - if langStr != "" { - l, err := syntax.ParseLanguage(langStr) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid Language for 'lang': %s", err), - }) - } - params.Lang = &l - } - // TODO: could be multiple tag params; guess we should "bind"? - tags := e.Request().URL.Query()["tags"] - if len(tags) > 0 { - params.Tags = tags - } - - offset, limit, err := parseCursorLimit(e) - if err != nil { - span.SetAttributes(attribute.String("error", fmt.Sprintf("invalid cursor/limit: %s", err))) - span.SetStatus(codes.Error, err.Error()) - return err - } - - params.Offset = offset - params.Size = limit - span.SetAttributes(attribute.Int("offset", offset), attribute.Int("limit", limit)) - - out, err := s.SearchPosts(ctx, ¶ms) - if err != nil { - span.SetAttributes(attribute.String("error", fmt.Sprintf("failed to SearchPosts: %s", err))) - span.SetStatus(codes.Error, err.Error()) - return err - } - - span.SetAttributes(attribute.Int("posts.length", len(out.Posts))) - - return e.JSON(200, out) -} - -func (s *Server) handleSearchActorsSkeleton(e echo.Context) error { - ctx, span := tracer.Start(e.Request().Context(), "handleSearchActorsSkeleton") - defer span.End() - - span.SetAttributes(attribute.String("query", e.QueryParam("q"))) - - q := strings.TrimSpace(e.QueryParam("q")) - if q == "" { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": "must pass non-empty search query", - }) - } - - offset, limit, err := parseCursorLimit(e) - if err != nil { - span.SetAttributes(attribute.String("error", fmt.Sprintf("invalid cursor/limit: %s", err))) - span.SetStatus(codes.Error, err.Error()) - return err - } - - typeahead := false - if q := strings.TrimSpace(e.QueryParam("typeahead")); q == "true" || q == "1" || q == "y" { - typeahead = true - } - - params := ActorSearchParams{ - Query: q, - Typeahead: typeahead, - Offset: offset, - Size: limit, - } - - viewerStr := e.QueryParam("viewer") - if viewerStr != "" { - d, err := syntax.ParseDID(viewerStr) - if err != nil { - return e.JSON(400, map[string]any{ - "error": "BadRequest", - "message": fmt.Sprintf("invalid DID for 'viewer': %s", err), - }) - } - params.Viewer = &d - } - - span.SetAttributes( - attribute.Int("offset", offset), - attribute.Int("limit", limit), - attribute.Bool("typeahead", typeahead), - ) - - out, err := s.SearchProfiles(ctx, ¶ms) - if err != nil { - span.SetAttributes(attribute.String("error", fmt.Sprintf("failed to SearchProfiles: %s", err))) - span.SetStatus(codes.Error, err.Error()) - return err - } - - span.SetAttributes(attribute.Int("actors.length", len(out.Actors))) - - return e.JSON(200, out) -} - -func (s *Server) SearchPosts(ctx context.Context, params *PostSearchParams) (*appbsky.UnspeccedSearchPostsSkeleton_Output, error) { - ctx, span := tracer.Start(ctx, "SearchPosts") - defer span.End() - - resp, err := DoSearchPosts(ctx, s.dir, s.escli, s.postIndex, params) - if err != nil { - return nil, err - } - - posts := []*appbsky.UnspeccedDefs_SkeletonSearchPost{} - for _, r := range resp.Hits.Hits { - var doc PostDoc - if err := json.Unmarshal(r.Source, &doc); err != nil { - return nil, fmt.Errorf("decoding post doc from search response: %w", err) - } - - did, err := syntax.ParseDID(doc.DID) - if err != nil { - return nil, fmt.Errorf("invalid DID in indexed document: %w", err) - } - - posts = append(posts, &appbsky.UnspeccedDefs_SkeletonSearchPost{ - Uri: fmt.Sprintf("at://%s/app.bsky.feed.post/%s", did, doc.RecordRkey), - }) - } - - out := appbsky.UnspeccedSearchPostsSkeleton_Output{Posts: posts} - if len(posts) == params.Size && (params.Offset+params.Size) < 10000 { - s := fmt.Sprintf("%d", params.Offset+params.Size) - out.Cursor = &s - } - if resp.Hits.Total.Relation == "eq" { - i := int64(resp.Hits.Total.Value) - out.HitsTotal = &i - } - return &out, nil -} - -func (s *Server) SearchProfiles(ctx context.Context, params *ActorSearchParams) (*appbsky.UnspeccedSearchActorsSkeleton_Output, error) { - ctx, span := tracer.Start(ctx, "SearchProfiles") - defer span.End() - span.SetAttributes( - attribute.String("query", params.Query), - attribute.Bool("typeahead", params.Typeahead), - attribute.Int("offset", params.Offset), - attribute.Int("size", params.Size), - ) - - var globalResp *EsSearchResponse - var personalizedResp *EsSearchResponse - var globalErr error - var personalizedErr error - - wg := sync.WaitGroup{} - - wg.Add(1) - // Conduct the global search - go func(myQ ActorSearchParams) { - defer wg.Done() - // Clear out the following list to conduct the global search - myQ.Follows = nil - - if myQ.Typeahead { - globalResp, globalErr = DoSearchProfilesTypeahead(ctx, s.escli, s.profileIndex, &myQ) - } else { - globalResp, globalErr = DoSearchProfiles(ctx, s.dir, s.escli, s.profileIndex, &myQ) - } - }(*params) - - // If we have a following list, conduct a second search to filter the results - if len(params.Follows) > 0 { - wg.Add(1) - go func(myQ ActorSearchParams) { - defer wg.Done() - if myQ.Typeahead { - personalizedResp, personalizedErr = DoSearchProfilesTypeahead(ctx, s.escli, s.profileIndex, &myQ) - } else { - personalizedResp, personalizedErr = DoSearchProfiles(ctx, s.dir, s.escli, s.profileIndex, &myQ) - } - }(*params) - } - - wg.Wait() - - if globalErr != nil { - return nil, globalErr - } - - if len(params.Follows) > 0 { - if personalizedErr != nil { - return nil, personalizedErr - } - - followingBoost := 0.1 - - // Insert the personalized results into the global results, deduping as we go and maintaining score-order - followingSeen := map[string]struct{}{} - for _, r := range personalizedResp.Hits.Hits { - var doc ProfileDoc - if err := json.Unmarshal(r.Source, &doc); err != nil { - return nil, fmt.Errorf("decoding profile doc from search response: %w", err) - } - - did, err := syntax.ParseDID(doc.DID) - if err != nil { - return nil, fmt.Errorf("invalid DID in indexed document: %w", err) - } - - if _, ok := followingSeen[did.String()]; ok { - continue - } - - followingSeen[did.String()] = struct{}{} - - // Insert the profile into the global results - globalResp.Hits.Hits = append(globalResp.Hits.Hits, r) - } - - // Walk the combined results and boost the scores of the personalized results and dedupe - seen := map[string]struct{}{} - deduped := []EsSearchHit{} - for _, r := range globalResp.Hits.Hits { - var doc ProfileDoc - if err := json.Unmarshal(r.Source, &doc); err != nil { - return nil, fmt.Errorf("decoding profile doc from search response: %w", err) - } - - did, err := syntax.ParseDID(doc.DID) - if err != nil { - return nil, fmt.Errorf("invalid DID in indexed document: %w", err) - } - - // Boost the score of the personalized results - if _, ok := followingSeen[did.String()]; ok { - r.Score += followingBoost - } - - // Dedupe the results - if _, ok := seen[did.String()]; ok { - continue - } - - seen[did.String()] = struct{}{} - deduped = append(deduped, r) - } - - // Sort the results by score - slices.SortFunc(deduped, func(a, b EsSearchHit) int { - if a.Score < b.Score { - return 1 - } - if a.Score > b.Score { - return -1 - } - return 0 - }) - - // Trim the results to the requested size - if len(deduped) > params.Size { - deduped = deduped[:params.Size] - } - - globalResp.Hits.Hits = deduped - } - - actors := []*appbsky.UnspeccedDefs_SkeletonSearchActor{} - for _, r := range globalResp.Hits.Hits { - var doc ProfileDoc - if err := json.Unmarshal(r.Source, &doc); err != nil { - return nil, fmt.Errorf("decoding profile doc from search response: %w", err) - } - - did, err := syntax.ParseDID(doc.DID) - if err != nil { - return nil, fmt.Errorf("invalid DID in indexed document: %w", err) - } - - actors = append(actors, &appbsky.UnspeccedDefs_SkeletonSearchActor{ - Did: did.String(), - }) - } - - out := appbsky.UnspeccedSearchActorsSkeleton_Output{Actors: actors} - if len(actors) == params.Size && (params.Offset+params.Size) < 10000 { - s := fmt.Sprintf("%d", params.Offset+params.Size) - out.Cursor = &s - } - if globalResp.Hits.Total.Relation == "eq" { - i := int64(globalResp.Hits.Total.Value) - out.HitsTotal = &i - } - return &out, nil -} diff --git a/search/indexing.go b/search/indexing.go deleted file mode 100644 index 4bdc130a..00000000 --- a/search/indexing.go +++ /dev/null @@ -1,606 +0,0 @@ -package search - -import ( - "bytes" - "context" - _ "embed" - "encoding/json" - "fmt" - "io" - "log/slog" - "os" - "strings" - "time" - - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - "github.com/bluesky-social/indigo/backfill" - "github.com/bluesky-social/indigo/xrpc" - "github.com/ipfs/go-cid" - "github.com/labstack/echo/v4" - "go.opentelemetry.io/otel/attribute" - "golang.org/x/time/rate" - gorm "gorm.io/gorm" - - es "github.com/opensearch-project/opensearch-go/v2" - esapi "github.com/opensearch-project/opensearch-go/v2/opensearchapi" -) - -type Indexer struct { - escli *es.Client - postIndex string - profileIndex string - db *gorm.DB - relayhost string - relayXRPC *xrpc.Client - dir identity.Directory - echo *echo.Echo - logger *slog.Logger - - bfs *backfill.Gormstore - bf *backfill.Backfiller - - enableRepoDiscovery bool - - indexLimiter *rate.Limiter - profileQueue chan *ProfileIndexJob - postQueue chan *PostIndexJob - pagerankQueue chan *PagerankIndexJob -} - -type IndexerConfig struct { - RelayHost string - ProfileIndex string - PostIndex string - Logger *slog.Logger - RelaySyncRateLimit int - IndexMaxConcurrency int - DiscoverRepos bool - IndexingRateLimit int -} - -type ProfileIndexJob struct { - ident *identity.Identity - record *appbsky.ActorProfile - rcid cid.Cid -} - -type PostIndexJob struct { - did syntax.DID - record *appbsky.FeedPost - rcid cid.Cid - rkey string -} - -type PagerankIndexJob struct { - did syntax.DID - rank float64 -} - -func NewIndexer(db *gorm.DB, escli *es.Client, dir identity.Directory, config IndexerConfig) (*Indexer, error) { - logger := config.Logger - if logger == nil { - logger = slog.New(slog.NewJSONHandler(os.Stdout, &slog.HandlerOptions{ - Level: slog.LevelInfo, - })) - } - logger = logger.With("component", "indexer") - - logger.Info("running database migrations") - db.AutoMigrate(&LastSeq{}) - db.AutoMigrate(&backfill.GormDBJob{}) - - relayWS := config.RelayHost - if !strings.HasPrefix(relayWS, "ws") { - return nil, fmt.Errorf("specified bgs host must include 'ws://' or 'wss://'") - } - - relayHTTP := strings.Replace(relayWS, "ws", "http", 1) - relayXRPC := &xrpc.Client{ - Host: relayHTTP, - } - - limiter := rate.NewLimiter(rate.Limit(config.IndexingRateLimit), 10_000) - - idx := &Indexer{ - escli: escli, - profileIndex: config.ProfileIndex, - postIndex: config.PostIndex, - db: db, - relayhost: config.RelayHost, - relayXRPC: relayXRPC, - dir: dir, - logger: logger, - enableRepoDiscovery: config.DiscoverRepos, - - indexLimiter: limiter, - profileQueue: make(chan *ProfileIndexJob, 1000), - postQueue: make(chan *PostIndexJob, 1000), - pagerankQueue: make(chan *PagerankIndexJob, 1000), - } - - bfstore := backfill.NewGormstore(db) - opts := backfill.DefaultBackfillOptions() - - if config.RelaySyncRateLimit > 0 { - opts.SyncRequestsPerSecond = config.RelaySyncRateLimit - opts.ParallelBackfills = 2 * config.RelaySyncRateLimit - } else { - opts.SyncRequestsPerSecond = 8 - } - - opts.RelayHost = relayHTTP - if config.IndexMaxConcurrency > 0 { - opts.ParallelRecordCreates = config.IndexMaxConcurrency - } else { - opts.ParallelRecordCreates = 20 - } - opts.NSIDFilter = "app.bsky." - bf := backfill.NewBackfiller( - "search", - bfstore, - idx.handleCreateOrUpdate, - idx.handleCreateOrUpdate, - idx.handleDelete, - opts, - ) - // reuse identity directory (for efficient caching) - bf.Directory = dir - - idx.bfs = bfstore - idx.bf = bf - - return idx, nil -} - -//go:embed post_schema.json -var palomarPostSchemaJSON string - -//go:embed profile_schema.json -var palomarProfileSchemaJSON string - -func (idx *Indexer) EnsureIndices(ctx context.Context) error { - indices := []struct { - Name string - SchemaJSON string - }{ - {Name: idx.postIndex, SchemaJSON: palomarPostSchemaJSON}, - {Name: idx.profileIndex, SchemaJSON: palomarProfileSchemaJSON}, - } - for _, index := range indices { - resp, err := idx.escli.Indices.Exists([]string{index.Name}) - if err != nil { - return err - } - defer resp.Body.Close() - io.ReadAll(resp.Body) - if resp.IsError() && resp.StatusCode != 404 { - return fmt.Errorf("failed to check index existence") - } - if resp.StatusCode == 404 { - idx.logger.Warn("creating opensearch index", "index", index.Name) - if len(index.SchemaJSON) < 2 { - return fmt.Errorf("empty schema file (go:embed failed)") - } - buf := strings.NewReader(index.SchemaJSON) - resp, err := idx.escli.Indices.Create( - index.Name, - idx.escli.Indices.Create.WithBody(buf)) - if err != nil { - return err - } - defer resp.Body.Close() - io.ReadAll(resp.Body) - if resp.IsError() { - return fmt.Errorf("failed to create index") - } - } - } - return nil -} - -func (idx *Indexer) runPostIndexer(ctx context.Context) { - ctx, span := tracer.Start(ctx, "runPostIndexer") - defer span.End() - - // Batch up to 1000 posts at a time, or every 5 seconds - tick := time.NewTicker(5 * time.Second) - defer tick.Stop() - - var posts []*PostIndexJob - for { - select { - case <-ctx.Done(): - return - case <-tick.C: - if len(posts) > 0 { - err := idx.indexLimiter.WaitN(ctx, len(posts)) - if err != nil { - idx.logger.Error("failed to wait for rate limiter", "err", err) - continue - } - err = idx.indexPosts(ctx, posts) - if err != nil { - idx.logger.Error("failed to index posts", "err", err) - } - posts = posts[:0] - } - case job := <-idx.postQueue: - posts = append(posts, job) - if len(posts) >= 1000 { - err := idx.indexLimiter.WaitN(ctx, len(posts)) - if err != nil { - idx.logger.Error("failed to wait for rate limiter", "err", err) - continue - } - err = idx.indexPosts(ctx, posts) - if err != nil { - idx.logger.Error("failed to index posts", "err", err) - } - posts = posts[:0] - } - } - } -} - -func (idx *Indexer) runProfileIndexer(ctx context.Context) { - ctx, span := tracer.Start(ctx, "runProfileIndexer") - defer span.End() - - // Batch up to 1000 profiles at a time, or every 5 seconds - tick := time.NewTicker(5 * time.Second) - defer tick.Stop() - - var profiles []*ProfileIndexJob - for { - select { - case <-ctx.Done(): - return - case <-tick.C: - if len(profiles) > 0 { - err := idx.indexLimiter.WaitN(ctx, len(profiles)) - if err != nil { - idx.logger.Error("failed to wait for rate limiter", "err", err) - continue - } - err = idx.indexProfiles(ctx, profiles) - if err != nil { - idx.logger.Error("failed to index profiles", "err", err) - } - profiles = profiles[:0] - } - case job := <-idx.profileQueue: - profiles = append(profiles, job) - if len(profiles) >= 1000 { - err := idx.indexLimiter.WaitN(ctx, len(profiles)) - if err != nil { - idx.logger.Error("failed to wait for rate limiter", "err", err) - continue - } - err = idx.indexProfiles(ctx, profiles) - if err != nil { - idx.logger.Error("failed to index profiles", "err", err) - } - profiles = profiles[:0] - } - } - } -} - -func (idx *Indexer) runPagerankIndexer(ctx context.Context) { - ctx, span := tracer.Start(ctx, "runPagerankIndexer") - defer span.End() - - // Batch up to 1000 pageranks at a time, or every 5 seconds - tick := time.NewTicker(5 * time.Second) - defer tick.Stop() - - var pageranks []*PagerankIndexJob - for { - select { - case <-ctx.Done(): - return - case <-tick.C: - if len(pageranks) > 0 { - err := idx.indexLimiter.WaitN(ctx, len(pageranks)) - if err != nil { - idx.logger.Error("failed to wait for rate limiter", "err", err) - continue - } - err = idx.indexPageranks(ctx, pageranks) - if err != nil { - idx.logger.Error("failed to index pageranks", "err", err) - } - pageranks = pageranks[:0] - } - case job := <-idx.pagerankQueue: - pageranks = append(pageranks, job) - if len(pageranks) >= 1000 { - err := idx.indexLimiter.WaitN(ctx, len(pageranks)) - if err != nil { - idx.logger.Error("failed to wait for rate limiter", "err", err) - continue - } - err = idx.indexPageranks(ctx, pageranks) - if err != nil { - idx.logger.Error("failed to index pageranks", "err", err) - } - pageranks = pageranks[:0] - } - } - } -} - -func (idx *Indexer) deletePost(ctx context.Context, did syntax.DID, recordPath string) error { - ctx, span := tracer.Start(ctx, "deletePost") - defer span.End() - span.SetAttributes(attribute.String("repo", did.String()), attribute.String("path", recordPath)) - - logger := idx.logger.With("repo", did, "path", recordPath, "op", "deletePost") - - parts := strings.SplitN(recordPath, "/", 3) - if len(parts) < 2 { - logger.Warn("skipping post record with malformed path") - return nil - } - rkey, err := syntax.ParseTID(parts[1]) - if err != nil { - logger.Warn("skipping post record with non-TID rkey") - return nil - } - - docID := fmt.Sprintf("%s_%s", did.String(), rkey) - logger.Info("deleting post from index", "docID", docID) - req := esapi.DeleteRequest{ - Index: idx.postIndex, - DocumentID: docID, - Refresh: "true", - } - - err = idx.indexLimiter.Wait(ctx) - if err != nil { - logger.Warn("failed to wait for rate limiter", "err", err) - return err - } - res, err := req.Do(ctx, idx.escli) - if err != nil { - return fmt.Errorf("failed to delete post: %w", err) - } - defer res.Body.Close() - body, err := io.ReadAll(res.Body) - if err != nil { - return fmt.Errorf("failed to read indexing response: %w", err) - } - if res.IsError() { - logger.Warn("opensearch indexing error", "status_code", res.StatusCode, "response", res, "body", string(body)) - return fmt.Errorf("indexing error, code=%d", res.StatusCode) - } - return nil -} - -func (idx *Indexer) indexPosts(ctx context.Context, jobs []*PostIndexJob) error { - ctx, span := tracer.Start(ctx, "indexPosts") - defer span.End() - span.SetAttributes(attribute.Int("num_posts", len(jobs))) - - log := idx.logger.With("op", "indexPosts") - start := time.Now() - - var buf bytes.Buffer - for i := range jobs { - job := jobs[i] - doc := TransformPost(job.record, job.did, job.rkey, job.rcid.String()) - docBytes, err := json.Marshal(doc) - if err != nil { - log.Warn("failed to marshal post", "err", err) - return err - } - - indexScript := []byte(fmt.Sprintf(`{"index":{"_id":"%s"}}%s`, doc.DocId(), "\n")) - docBytes = append(docBytes, "\n"...) - - buf.Grow(len(indexScript) + len(docBytes)) - buf.Write(indexScript) - buf.Write(docBytes) - } - - log.Info("indexing posts", "num_posts", len(jobs)) - - res, err := idx.escli.Bulk(bytes.NewReader(buf.Bytes()), idx.escli.Bulk.WithIndex(idx.postIndex)) - if err != nil { - log.Warn("failed to send bulk indexing request", "err", err) - return fmt.Errorf("failed to send bulk indexing request: %w", err) - } - defer res.Body.Close() - - if res.IsError() { - body, err := io.ReadAll(res.Body) - if err != nil { - log.Warn("failed to read bulk indexing response", "err", err) - return fmt.Errorf("failed to read bulk indexing response: %w", err) - } - log.Warn("opensearch bulk indexing error", "status_code", res.StatusCode, "response", res, "body", string(body)) - return fmt.Errorf("bulk indexing error, code=%d", res.StatusCode) - } - - log.Info("indexed posts", "num_posts", len(jobs), "duration", time.Since(start)) - - return nil -} - -func (idx *Indexer) indexProfiles(ctx context.Context, jobs []*ProfileIndexJob) error { - ctx, span := tracer.Start(ctx, "indexProfiles") - defer span.End() - span.SetAttributes(attribute.Int("num_profiles", len(jobs))) - - log := idx.logger.With("op", "indexProfiles") - start := time.Now() - - var buf bytes.Buffer - for i := range jobs { - job := jobs[i] - - doc := TransformProfile(job.record, job.ident, job.rcid.String()) - docBytes, err := json.Marshal(doc) - if err != nil { - log.Warn("failed to marshal profile", "err", err) - return err - } - - indexScript := []byte(fmt.Sprintf(`{"index":{"_id":"%s"}}%s`, job.ident.DID.String(), "\n")) - docBytes = append(docBytes, "\n"...) - - buf.Grow(len(indexScript) + len(docBytes)) - buf.Write(indexScript) - buf.Write(docBytes) - } - - log.Info("indexing profiles", "num_profiles", len(jobs)) - - res, err := idx.escli.Bulk(bytes.NewReader(buf.Bytes()), idx.escli.Bulk.WithIndex(idx.profileIndex)) - if err != nil { - log.Warn("failed to send bulk indexing request", "err", err) - return fmt.Errorf("failed to send bulk indexing request: %w", err) - } - defer res.Body.Close() - - if res.IsError() { - body, err := io.ReadAll(res.Body) - if err != nil { - log.Warn("failed to read bulk indexing response", "err", err) - return fmt.Errorf("failed to read bulk indexing response: %w", err) - } - log.Warn("opensearch bulk indexing error", "status_code", res.StatusCode, "response", res, "body", string(body)) - return fmt.Errorf("bulk indexing error, code=%d", res.StatusCode) - } - - log.Info("indexed profiles", "num_profiles", len(jobs), "duration", time.Since(start)) - - return nil -} - -// indexPageranks uses the OpenSearch bulk API to update the pageranks for the given DIDs -func (idx *Indexer) indexPageranks(ctx context.Context, pageranks []*PagerankIndexJob) error { - ctx, span := tracer.Start(ctx, "indexPageranks") - defer span.End() - span.SetAttributes(attribute.Int("num_profiles", len(pageranks))) - - log := idx.logger.With("op", "indexPageranks") - - log.Info("updating profile pageranks") - - var buf bytes.Buffer - for _, pr := range pageranks { - updateScript := map[string]any{ - "script": map[string]any{ - "source": "ctx._source.pagerank = params.pagerank", - "lang": "painless", - "params": map[string]any{ - "pagerank": pr.rank, - }, - }, - } - updateScriptJSON, err := json.Marshal(updateScript) - if err != nil { - log.Warn("failed to marshal update script", "err", err) - return err - } - - updateMetaJSON := []byte(fmt.Sprintf(`{"update":{"_id":"%s"}}%s`, pr.did.String(), "\n")) - updateScriptJSON = append(updateScriptJSON, "\n"...) - - buf.Grow(len(updateMetaJSON) + len(updateScriptJSON)) - buf.Write(updateMetaJSON) - buf.Write(updateScriptJSON) - } - - res, err := idx.escli.Bulk(bytes.NewReader(buf.Bytes()), idx.escli.Bulk.WithIndex(idx.profileIndex)) - if err != nil { - log.Warn("failed to send bulk indexing request", "err", err) - return fmt.Errorf("failed to send bulk indexing request: %w", err) - } - defer res.Body.Close() - - if res.IsError() { - body, err := io.ReadAll(res.Body) - if err != nil { - log.Warn("failed to read bulk indexing response", "err", err) - return fmt.Errorf("failed to read bulk indexing response: %w", err) - } - log.Warn("opensearch bulk indexing error", "status_code", res.StatusCode, "response", res, "body", string(body)) - return fmt.Errorf("bulk indexing error, code=%d", res.StatusCode) - } - - return nil -} - -func (idx *Indexer) updateUserHandle(ctx context.Context, did syntax.DID, handle string) error { - ctx, span := tracer.Start(ctx, "updateUserHandle") - defer span.End() - span.SetAttributes(attribute.String("repo", did.String()), attribute.String("event.handle", handle)) - - log := idx.logger.With("repo", did.String(), "op", "updateUserHandle", "handle_from_event", handle) - - err := idx.dir.Purge(ctx, did.AtIdentifier()) - if err != nil { - log.Warn("failed to purge DID from directory", "err", err) - return err - } - - ident, err := idx.dir.LookupDID(ctx, did) - if err != nil { - log.Warn("failed to lookup DID in directory", "err", err) - return err - } - - if ident == nil { - log.Warn("got nil identity from directory") - return fmt.Errorf("got nil identity from directory") - } - - log.Info("updating user handle", "handle_from_dir", ident.Handle) - span.SetAttributes(attribute.String("dir.handle", ident.Handle.String())) - - b, err := json.Marshal(map[string]any{ - "script": map[string]any{ - "source": "ctx._source.handle = params.handle", - "lang": "painless", - "params": map[string]any{ - "handle": ident.Handle, - }, - }, - }) - if err != nil { - log.Warn("failed to marshal update script", "err", err) - return err - } - - req := esapi.UpdateRequest{ - Index: idx.profileIndex, - DocumentID: did.String(), - Body: bytes.NewReader(b), - } - - err = idx.indexLimiter.Wait(ctx) - if err != nil { - log.Warn("failed to wait for rate limiter", "err", err) - return err - } - res, err := req.Do(ctx, idx.escli) - if err != nil { - log.Warn("failed to send indexing request", "err", err) - return fmt.Errorf("failed to send indexing request: %w", err) - } - defer res.Body.Close() - body, err := io.ReadAll(res.Body) - if err != nil { - log.Warn("failed to read indexing response", "err", err) - return fmt.Errorf("failed to read indexing response: %w", err) - } - if res.IsError() { - log.Warn("opensearch indexing error", "status_code", res.StatusCode, "response", res, "body", string(body)) - return fmt.Errorf("indexing error, code=%d", res.StatusCode) - } - return nil -} diff --git a/search/japanese.go b/search/japanese.go deleted file mode 100644 index 9784bf10..00000000 --- a/search/japanese.go +++ /dev/null @@ -1,14 +0,0 @@ -package search - -import ( - "regexp" -) - -// U+3040 - U+30FF: hiragana and katakana (Japanese only) -// U+FF66 - U+FF9F: half-width katakana (Japanese only) -var japaneseRegex = regexp.MustCompile(`[\x{3040}-\x{30ff}\x{ff66}-\x{ff9f}]`) - -// helper to check if an input string contains any Japanese-specific characters (hiragana or katakana). will not trigger on CJK characters which are not specific to Japanese -func containsJapanese(text string) bool { - return japaneseRegex.MatchString(text) -} diff --git a/search/japanese_test.go b/search/japanese_test.go deleted file mode 100644 index df7606ad..00000000 --- a/search/japanese_test.go +++ /dev/null @@ -1,23 +0,0 @@ -package search - -import ( - "testing" - - "github.com/stretchr/testify/assert" -) - -func TestJapaneseDetection(t *testing.T) { - assert := assert.New(t) - - assert.False(containsJapanese("")) - assert.False(containsJapanese("basic english")) - assert.False(containsJapanese("basic english")) - - assert.True(containsJapanese("学校から帰って熱いお風呂に入ったら力一杯がんばる")) - assert.True(containsJapanese("パリ")) - assert.True(containsJapanese("ハリー・ポッター")) - assert.True(containsJapanese("some japanese パリ and some english")) - - // CJK, but not japanese-specific - assert.False(containsJapanese("熱力学")) -} diff --git a/search/metrics.go b/search/metrics.go deleted file mode 100644 index 9378af3b..00000000 --- a/search/metrics.go +++ /dev/null @@ -1,158 +0,0 @@ -package search - -import ( - "errors" - "net/http" - "strconv" - "strings" - "time" - - "github.com/labstack/echo/v4" - "github.com/prometheus/client_golang/prometheus" - "github.com/prometheus/client_golang/prometheus/promauto" -) - -var postsReceived = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_posts_received", - Help: "Number of posts received", -}) - -var postsIndexed = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_posts_indexed", - Help: "Number of posts indexed", -}) - -var postsFailed = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_posts_failed", - Help: "Number of posts that failed indexing", -}) - -var postsDeleted = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_posts_deleted", - Help: "Number of posts deleted", -}) - -var profilesReceived = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_profiles_received", - Help: "Number of profiles received", -}) - -var profilesIndexed = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_profiles_indexed", - Help: "Number of profiles indexed", -}) - -var profilesFailed = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_profiles_failed", - Help: "Number of profiles that failed indexing", -}) - -var profilesDeleted = promauto.NewCounter(prometheus.CounterOpts{ - Name: "search_profiles_deleted", - Help: "Number of profiles deleted", -}) - -var currentSeq = promauto.NewGauge(prometheus.GaugeOpts{ - Name: "search_current_seq", - Help: "Current sequence number", -}) - -var reqSz = promauto.NewHistogramVec(prometheus.HistogramOpts{ - Name: "http_request_size_bytes", - Help: "A histogram of request sizes for requests.", - Buckets: prometheus.ExponentialBuckets(100, 10, 8), -}, []string{"code", "method", "path", "extras"}) - -var reqDur = promauto.NewHistogramVec(prometheus.HistogramOpts{ - Name: "http_request_duration_seconds", - Help: "A histogram of latencies for requests.", - Buckets: prometheus.ExponentialBuckets(0.0001, 2, 18), -}, []string{"code", "method", "path", "extras"}) - -var reqCnt = promauto.NewCounterVec(prometheus.CounterOpts{ - Name: "http_requests_total", - Help: "A counter for requests to the wrapped handler.", -}, []string{"code", "method", "path", "extras"}) - -var resSz = promauto.NewHistogramVec(prometheus.HistogramOpts{ - Name: "http_response_size_bytes", - Help: "A histogram of response sizes for requests.", - Buckets: prometheus.ExponentialBuckets(100, 10, 8), -}, []string{"code", "method", "path", "extras"}) - -// MetricsMiddleware defines handler function for metrics middleware -func MetricsMiddleware(next echo.HandlerFunc) echo.HandlerFunc { - return func(c echo.Context) error { - path := c.Path() - if path == "/metrics" || path == "/_health" { - return next(c) - } - - start := time.Now() - requestSize := computeApproximateRequestSize(c.Request()) - - err := next(c) - - status := c.Response().Status - if err != nil { - var httpError *echo.HTTPError - if errors.As(err, &httpError) { - status = httpError.Code - } - if status == 0 || status == http.StatusOK { - status = http.StatusInternalServerError - } - } - - elapsed := float64(time.Since(start)) / float64(time.Second) - - statusStr := strconv.Itoa(status) - method := c.Request().Method - - responseSize := float64(c.Response().Size) - - // Custom label for Typeahead search queries - typeahead := false - if q := strings.TrimSpace(c.QueryParam("typeahead")); q == "true" || q == "1" || q == "y" { - typeahead = true - } - - labels := []string{statusStr, method, path} - if typeahead { - labels = append(labels, "typeahead") - } else { - labels = append(labels, "_none") - } - - reqDur.WithLabelValues(labels...).Observe(elapsed) - reqCnt.WithLabelValues(labels...).Inc() - reqSz.WithLabelValues(labels...).Observe(float64(requestSize)) - resSz.WithLabelValues(labels...).Observe(responseSize) - - return err - } -} - -func computeApproximateRequestSize(r *http.Request) int { - s := 0 - if r.URL != nil { - s = len(r.URL.Path) - } - - s += len(r.Method) - s += len(r.Proto) - for name, values := range r.Header { - s += len(name) - for _, value := range values { - s += len(value) - } - } - s += len(r.Host) - - // N.B. r.Form and r.MultipartForm are assumed to be included in r.URL. - - if r.ContentLength != -1 { - s += int(r.ContentLength) - } - return s -} diff --git a/search/parse_query.go b/search/parse_query.go deleted file mode 100644 index f037565d..00000000 --- a/search/parse_query.go +++ /dev/null @@ -1,151 +0,0 @@ -package search - -import ( - "context" - "log/slog" - "strings" - "time" - - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" -) - -// ParsePostQuery takes a query string and pulls out some facet patterns ("from:handle.net") as filters -func ParsePostQuery(ctx context.Context, dir identity.Directory, raw string, viewer *syntax.DID) PostSearchParams { - quoted := false - parts := strings.FieldsFunc(raw, func(r rune) bool { - if r == '"' { - quoted = !quoted - } - return r == ' ' && !quoted - }) - - params := PostSearchParams{} - - keep := make([]string, 0, len(parts)) - for _, p := range parts { - // pass-through quoted, either phrase or single token - if strings.HasPrefix(p, "\"") { - keep = append(keep, p) - continue - } - - // tags (array) - if strings.HasPrefix(p, "#") && len(p) > 1 { - params.Tags = append(params.Tags, p[1:]) - continue - } - - // handle (mention) - if strings.HasPrefix(p, "@") && len(p) > 1 { - handle, err := syntax.ParseHandle(p[1:]) - if err != nil { - keep = append(keep, p) - continue - } - id, err := dir.LookupHandle(ctx, handle) - if err != nil { - if err != identity.ErrHandleNotFound { - slog.Error("failed to resolve handle", "err", err) - } - continue - } - params.Mentions = &id.DID - continue - } - - tokParts := strings.SplitN(p, ":", 2) - if len(tokParts) == 1 { - keep = append(keep, p) - continue - } - - switch tokParts[0] { - case "did": - // Used as a hack for `from:me` when supplied by the client - did, err := syntax.ParseDID(p) - if err != nil { - continue - } - params.Author = &did - continue - case "from", "to", "mentions": - raw := tokParts[1] - if raw == "me" { - if viewer != nil && tokParts[0] == "from" { - params.Author = viewer - } else if viewer != nil { - params.Mentions = viewer - } - continue - } - if strings.HasPrefix(raw, "@") && len(raw) > 1 { - raw = raw[1:] - } - handle, err := syntax.ParseHandle(raw) - if err != nil { - continue - } - id, err := dir.LookupHandle(ctx, handle) - if err != nil { - if err != identity.ErrHandleNotFound { - slog.Error("failed to resolve handle", "err", err) - } - continue - } - if tokParts[0] == "from" { - params.Author = &id.DID - } else { - params.Mentions = &id.DID - } - continue - case "http", "https": - params.URL = p - continue - case "domain": - params.Domain = tokParts[1] - continue - case "lang": - lang, err := syntax.ParseLanguage(tokParts[1]) - if nil == err { - params.Lang = &lang - } - continue - case "since", "until": - var dt syntax.Datetime - // first try just date - date, err := time.Parse(time.DateOnly, tokParts[1]) - if nil == err { - dt = syntax.Datetime(date.Format(syntax.AtprotoDatetimeLayout)) - } else { - // fallback to formal atproto datetime format - dt, err = syntax.ParseDatetimeLenient(tokParts[1]) - if err != nil { - continue - } - } - if tokParts[0] == "since" { - params.Since = &dt - } else { - params.Until = &dt - } - continue - } - - keep = append(keep, p) - } - - out := "" - for _, p := range keep { - if out == "" { - out = p - } else { - out += " " + p - } - } - if out == "" { - out = "*" - } - params.Query = out - return params -} diff --git a/search/parse_query_test.go b/search/parse_query_test.go deleted file mode 100644 index ca716304..00000000 --- a/search/parse_query_test.go +++ /dev/null @@ -1,98 +0,0 @@ -package search - -import ( - "context" - "testing" - - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - - "github.com/stretchr/testify/assert" -) - -func TestParseQuery(t *testing.T) { - ctx := context.Background() - assert := assert.New(t) - dir := identity.NewMockDirectory() - ident := identity.Identity{ - Handle: syntax.Handle("known.example.com"), - DID: syntax.DID("did:plc:abc222"), - } - dir.Insert(ident) - - var p PostSearchParams - - p = ParsePostQuery(ctx, dir, "", nil) - assert.Equal("*", p.Query) - assert.Empty(p.Filters()) - - q1 := "some +test \"with phrase\" -ok" - p = ParsePostQuery(ctx, dir, q1, nil) - assert.Equal(q1, p.Query) - assert.Empty(p.Filters()) - - q2 := "missing from:missing.example.com" - p = ParsePostQuery(ctx, dir, q2, nil) - assert.Equal("missing", p.Query) - assert.Empty(p.Filters()) - - q3 := "known from:known.example.com" - p = ParsePostQuery(ctx, dir, q3, nil) - assert.Equal("known", p.Query) - assert.NotNil(p.Author) - if p.Author != nil { - assert.Equal("did:plc:abc222", p.Author.String()) - } - - q4 := "from:known.example.com" - p = ParsePostQuery(ctx, dir, q4, nil) - assert.Equal("*", p.Query) - assert.Equal(1, len(p.Filters())) - - q5 := `from:known.example.com "multi word phrase" coolio blorg` - p = ParsePostQuery(ctx, dir, q5, nil) - assert.Equal(`"multi word phrase" coolio blorg`, p.Query) - assert.NotNil(p.Author) - if p.Author != nil { - assert.Equal("did:plc:abc222", p.Author.String()) - } - assert.Equal(1, len(p.Filters())) - - q6 := `from:known.example.com #cool_tag some other stuff` - p = ParsePostQuery(ctx, dir, q6, nil) - assert.Equal(`some other stuff`, p.Query) - assert.NotNil(p.Author) - if p.Author != nil { - assert.Equal("did:plc:abc222", p.Author.String()) - } - assert.Equal([]string{"cool_tag"}, p.Tags) - assert.Equal(2, len(p.Filters())) - - q7 := "known from:@known.example.com" - p = ParsePostQuery(ctx, dir, q7, nil) - assert.Equal("known", p.Query) - assert.NotNil(p.Author) - if p.Author != nil { - assert.Equal("did:plc:abc222", p.Author.String()) - } - assert.Equal(1, len(p.Filters())) - - q8 := "known from:me" - p = ParsePostQuery(ctx, dir, q8, &ident.DID) - assert.Equal("known", p.Query) - assert.NotNil(p.Author) - if p.Author != nil { - assert.Equal("did:plc:abc222", p.Author.String()) - } - assert.Equal(1, len(p.Filters())) - - q9 := "did:plc:abc222" - p = ParsePostQuery(ctx, dir, q9, nil) - assert.Equal("*", p.Query) - assert.Equal(1, len(p.Filters())) - if p.Author != nil { - assert.Equal("did:plc:abc222", p.Author.String()) - } - - // TODO: more parsing tests: bare handles, to:, since:, until:, URL, domain:, lang -} diff --git a/search/post_schema.json b/search/post_schema.json deleted file mode 100644 index e2bee31c..00000000 --- a/search/post_schema.json +++ /dev/null @@ -1,102 +0,0 @@ -{ -"settings": { - "index": { - "number_of_shards": 6, - "number_of_replicas": 1, - "refresh_interval": "5s", - "analysis": { - "analyzer": { - "default": { - "type": "custom", - "tokenizer": "standard", - "filter": [ "lowercase", "asciifolding" ] - }, - "textIcu": { - "type": "custom", - "tokenizer": "icu_tokenizer", - "char_filter": [ "icu_normalizer" ], - "filter": [ "icu_folding" ] - }, - "textIcuSearch": { - "type": "custom", - "tokenizer": "icu_tokenizer", - "char_filter": [ "icu_normalizer" ], - "filter": [ "icu_folding" ] - }, - "textJapanese": { - "type": "custom", - "tokenizer": "kuromoji_tokenizer", - "char_filter": [ "icu_normalizer" ], - "filter": [ - "kuromoji_baseform", - "kuromoji_part_of_speech", - "cjk_width", - "ja_stop", - "kuromoji_stemmer", - "lowercase" - ] - }, - "textJapaneseSearch": { - "type": "custom", - "tokenizer": "kuromoji_tokenizer", - "char_filter": [ "icu_normalizer" ], - "filter": [ - "kuromoji_baseform", - "kuromoji_part_of_speech", - "cjk_width", - "ja_stop", - "kuromoji_stemmer", - "lowercase" - ] - } - }, - "normalizer": { - "default": { - "type": "custom", - "char_filter": [], - "filter": ["lowercase"] - }, - "caseSensitive": { - "type": "custom", - "char_filter": [], - "filter": [] - } - } - } - } -}, -"mappings": { - "dynamic": false, - "properties": { - "doc_index_ts": { "type": "date" }, - "did": { "type": "keyword", "normalizer": "default", "doc_values": false }, - "record_rkey": { "type": "keyword", "normalizer": "default", "doc_values": false }, - "record_cid": { "type": "keyword", "normalizer": "default", "doc_values": false }, - - "created_at": { "type": "date" }, - "text": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch", "copy_to": "everything" }, - "text_ja": { "type": "text", "analyzer": "textJapanese", "search_analyzer": "textJapaneseSearch", "copy_to": "everything_ja" }, - "lang_code": { "type": "keyword", "normalizer": "default" }, - "lang_code_iso2": { "type": "keyword", "normalizer": "default" }, - "mention_did": { "type": "keyword", "normalizer": "default" }, - "embed_aturi": { "type": "keyword", "normalizer": "default" }, - "reply_root_aturi": { "type": "keyword", "normalizer": "default" }, - "embed_img_count": { "type": "integer" }, - "embed_img_alt_text": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch", "copy_to": "everything" }, - "embed_img_alt_text_ja": { "type": "text", "analyzer": "textJapanese", "search_analyzer": "textJapaneseSearch", "copy_to": "everything_ja" }, - "self_label": { "type": "keyword", "normalizer": "default" }, - - "url": { "type": "keyword", "normalizer": "default" }, - "domain": { "type": "keyword", "normalizer": "default" }, - "tag": { "type": "keyword", "normalizer": "default" }, - "emoji": { "type": "keyword", "normalizer": "caseSensitive" }, - - "likesFuzzy": { "type": "integer" }, - - "everything": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch" }, - "everything_ja": { "type": "text", "analyzer": "textJapanese", "search_analyzer": "textJapaneseSearch" }, - - "lang": { "type": "alias", "path": "lang_code_iso2" } - } -} -} diff --git a/search/profile_schema.json b/search/profile_schema.json deleted file mode 100644 index 9393e138..00000000 --- a/search/profile_schema.json +++ /dev/null @@ -1,70 +0,0 @@ -{ -"settings": { - "index": { - "number_of_shards": 1, - "number_of_replicas": 1, - "refresh_interval": "5s", - "analysis": { - "analyzer": { - "default": { - "type": "custom", - "tokenizer": "standard", - "filter": [ "lowercase", "asciifolding" ] - }, - "textIcu": { - "type": "custom", - "tokenizer": "icu_tokenizer", - "char_filter": [ "icu_normalizer" ], - "filter": [ "icu_folding" ] - }, - "textIcuSearch": { - "type": "custom", - "tokenizer": "icu_tokenizer", - "char_filter": [ "icu_normalizer" ], - "filter": [ "icu_folding" ] - } - }, - "normalizer": { - "default": { - "type": "custom", - "char_filter": [], - "filter": ["lowercase"] - }, - "caseSensitive": { - "type": "custom", - "char_filter": [], - "filter": [] - } - } - } - } -}, -"mappings": { - "dynamic": false, - "properties": { - "doc_index_ts": { "type": "date" }, - "did": { "type": "keyword", "normalizer": "default", "doc_values": false }, - "handle": { "type": "keyword", "normalizer": "default", "copy_to": ["everything", "typeahead"] }, - "record_cid": { "type": "keyword", "normalizer": "default", "doc_values": false }, - - "display_name": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch", "copy_to": ["everything", "typeahead"] }, - "description": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch", "copy_to": "everything" }, - "img_alt_text": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch", "copy_to": "everything" }, - "self_label": { "type": "keyword", "normalizer": "default" }, - - "url": { "type": "keyword", "normalizer": "default" }, - "domain": { "type": "keyword", "normalizer": "default" }, - "tag": { "type": "keyword", "normalizer": "default" }, - "emoji": { "type": "keyword", "normalizer": "caseSensitive" }, - - "has_avatar": { "type": "boolean" }, - "has_banner": { "type": "boolean" }, - - "pagerank": { "type": "float" }, - "followersFuzzy": { "type": "integer" }, - - "typeahead": { "type": "search_as_you_type", "analyzer": "textIcu", "search_analyzer": "textIcuSearch" }, - "everything": { "type": "text", "analyzer": "textIcu", "search_analyzer": "textIcuSearch" } - } -} -} diff --git a/search/query.go b/search/query.go deleted file mode 100644 index 52ac9b25..00000000 --- a/search/query.go +++ /dev/null @@ -1,435 +0,0 @@ -package search - -import ( - "bytes" - "context" - "encoding/json" - "fmt" - "io/ioutil" - "log/slog" - "strings" - - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - - es "github.com/opensearch-project/opensearch-go/v2" - "go.opentelemetry.io/otel/attribute" -) - -type EsSearchHit struct { - Index string `json:"_index"` - ID string `json:"_id"` - Score float64 `json:"_score"` - Source json.RawMessage `json:"_source"` -} - -type EsSearchHits struct { - Total struct { // not used - Value int - Relation string - } `json:"total"` - MaxScore float64 `json:"max_score"` - Hits []EsSearchHit `json:"hits"` -} - -type EsSearchResponse struct { - Took int `json:"took"` - TimedOut bool `json:"timed_out"` - Hits EsSearchHits `json:"hits"` -} - -type UserResult struct { - Did string `json:"did"` - Handle string `json:"handle"` -} - -type PostSearchResult struct { - Tid string `json:"tid"` - Cid string `json:"cid"` - User UserResult `json:"user"` - Post any `json:"post"` -} - -type PostSearchParams struct { - Query string `json:"q"` - Sort string `json:"sort"` - Author *syntax.DID `json:"author"` - Since *syntax.Datetime `json:"since"` - Until *syntax.Datetime `json:"until"` - Mentions *syntax.DID `json:"mentions"` - Lang *syntax.Language `json:"lang"` - Domain string `json:"domain"` - URL string `json:"url"` - Tags []string `json:"tag"` - Viewer *syntax.DID `json:"viewer"` - Offset int `json:"offset"` - Size int `json:"size"` -} - -type ActorSearchParams struct { - Query string `json:"q"` - Typeahead bool `json:"typeahead"` - Follows []syntax.DID `json:"follows"` - Viewer *syntax.DID `json:"viewer"` - Offset int `json:"offset"` - Size int `json:"size"` -} - -// Merges params from another param object in to this one. Intended to meld parsed query with HTTP query params, so not all functionality is supported, and priority is with the "current" object -func (p *PostSearchParams) Update(other *PostSearchParams) { - p.Query = other.Query - if p.Author == nil { - p.Author = other.Author - } - if p.Since == nil { - p.Since = other.Since - } - if p.Until == nil { - p.Until = other.Until - } - if p.Mentions == nil { - p.Mentions = other.Mentions - } - if p.Lang == nil { - p.Lang = other.Lang - } - if p.Domain == "" { - p.Domain = other.Domain - } - if p.URL == "" { - p.URL = other.URL - } - if len(p.Tags) == 0 { - p.Tags = other.Tags - } -} - -// Filters turns search params in to actual elasticsearch/opensearch filter DSL -func (p *PostSearchParams) Filters() []map[string]any { - var filters []map[string]any - - if p.Author != nil { - filters = append(filters, map[string]any{ - "term": map[string]any{"did": map[string]any{ - "value": p.Author.String(), - "case_insensitive": true, - }}, - }) - } - - if p.Mentions != nil { - filters = append(filters, map[string]any{ - "term": map[string]any{"mention_did": map[string]any{ - "value": p.Mentions.String(), - "case_insensitive": true, - }}, - }) - } - - if p.Lang != nil { - // TODO: extracting just the 2-char code would be good - filters = append(filters, map[string]any{ - "term": map[string]any{"lang_code_iso2": map[string]any{ - "value": p.Lang.String(), - "case_insensitive": true, - }}, - }) - } - - if p.Since != nil { - filters = append(filters, map[string]any{ - "range": map[string]any{ - "created_at": map[string]any{ - "gte": p.Since.String(), - }, - }, - }) - } - - if p.Until != nil { - filters = append(filters, map[string]any{ - "range": map[string]any{ - "created_at": map[string]any{ - "lt": p.Until.String(), - }, - }, - }) - } - - if p.URL != "" { - filters = append(filters, map[string]any{ - "term": map[string]any{"url": map[string]any{ - "value": NormalizeLossyURL(p.URL), - "case_insensitive": true, - }}, - }) - } - - if p.Domain != "" { - filters = append(filters, map[string]any{ - "term": map[string]any{"domain": map[string]any{ - "value": p.Domain, - "case_insensitive": true, - }}, - }) - } - - for _, tag := range p.Tags { - filters = append(filters, map[string]any{ - "term": map[string]any{ - "tag": map[string]any{ - "value": tag, - "case_insensitive": true, - }, - }, - }) - } - - return filters -} - -// Filters turns search params in to actual elasticsearch/opensearch filter DSL -func (p *ActorSearchParams) Filters() []map[string]any { - var filters []map[string]any - - if p.Follows != nil && len(p.Follows) > 0 { - follows := make([]string, len(p.Follows)) - for i, did := range p.Follows { - follows[i] = did.String() - } - filters = append(filters, map[string]any{ - "terms": map[string]any{ - "did": follows, - }, - }) - } - - return filters -} - -func checkParams(offset, size int) error { - if offset+size > 10000 || size > 250 || offset > 10000 || offset < 0 || size < 0 { - return fmt.Errorf("disallowed size/offset parameters") - } - return nil -} - -func DoSearchPosts(ctx context.Context, dir identity.Directory, escli *es.Client, index string, params *PostSearchParams) (*EsSearchResponse, error) { - ctx, span := tracer.Start(ctx, "DoSearchPosts") - defer span.End() - - if err := checkParams(params.Offset, params.Size); err != nil { - return nil, err - } - queryStringParams := ParsePostQuery(ctx, dir, params.Query, params.Viewer) - params.Update(&queryStringParams) - idx := "everything" - if containsJapanese(params.Query) { - idx = "everything_ja" - } - basic := map[string]any{ - "simple_query_string": map[string]any{ - "query": params.Query, - "fields": []string{idx}, - "flags": "AND|NOT|OR|PHRASE|PRECEDENCE|WHITESPACE", - "default_operator": "and", - "lenient": true, - "analyze_wildcard": false, - }, - } - filters := params.Filters() - // filter out future posts (TODO: temporary hack) - now := syntax.DatetimeNow() - filters = append(filters, map[string]any{ - "range": map[string]any{ - "created_at": map[string]any{ - "lte": now, - }, - }, - }) - query := map[string]any{ - "query": map[string]any{ - "bool": map[string]any{ - "must": basic, - "filter": filters, - }, - }, - "sort": map[string]any{ - "created_at": map[string]any{ - "order": "desc", - }, - }, - "size": params.Size, - "from": params.Offset, - } - - return doSearch(ctx, escli, index, query) -} - -func DoSearchProfiles(ctx context.Context, dir identity.Directory, escli *es.Client, index string, params *ActorSearchParams) (*EsSearchResponse, error) { - ctx, span := tracer.Start(ctx, "DoSearchProfiles") - defer span.End() - - if err := checkParams(params.Offset, params.Size); err != nil { - return nil, err - } - - filters := params.Filters() - - fulltext := map[string]any{ - "simple_query_string": map[string]any{ - "query": params.Query, - "fields": []string{"everything"}, - "flags": "AND|NOT|OR|PHRASE|PRECEDENCE|WHITESPACE", - "default_operator": "and", - "lenient": true, - "analyze_wildcard": false, - }, - } - primary := fulltext - - // if the query string is just a single token (after parsing out filter - // syntax), then have the primary query be an "OR" of the basic fulltext - // query and the typeahead query - if len(strings.Split(params.Query, " ")) == 1 { - typeahead := map[string]any{ - "multi_match": map[string]any{ - "query": params.Query, - "type": "bool_prefix", - "operator": "and", - "fields": []string{ - "typeahead", - "typeahead._2gram", - "typeahead._3gram", - }, - }, - } - primary = map[string]any{ - "bool": map[string]any{ - "should": []any{ - fulltext, - typeahead, - }, - }, - } - } - - query := map[string]any{ - "query": map[string]any{ - "bool": map[string]any{ - "must": primary, - "should": []any{ - map[string]any{"term": map[string]any{"has_avatar": true}}, - map[string]any{"term": map[string]any{"has_banner": true}}, - }, - "minimum_should_match": 0, - "boost": 0.5, - }, - }, - "size": params.Size, - "from": params.Offset, - } - - if len(filters) > 0 { - query["query"].(map[string]any)["bool"].(map[string]any)["filter"] = filters - } - - return doSearch(ctx, escli, index, query) -} - -func DoSearchProfilesTypeahead(ctx context.Context, escli *es.Client, index string, params *ActorSearchParams) (*EsSearchResponse, error) { - ctx, span := tracer.Start(ctx, "DoSearchProfilesTypeahead") - defer span.End() - - if err := checkParams(0, params.Size); err != nil { - return nil, err - } - - filters := params.Filters() - - query := map[string]any{ - "query": map[string]any{ - "bool": map[string]any{ - "must": map[string]any{ - "multi_match": map[string]any{ - "query": params.Query, - "type": "bool_prefix", - "operator": "and", - "fields": []string{ - "typeahead", - "typeahead._2gram", - "typeahead._3gram", - }, - }, - }, - }, - }, - "size": params.Size, - "from": params.Offset, - } - - if len(filters) > 0 { - query["query"].(map[string]any)["bool"].(map[string]any)["filter"] = filters - } - - return doSearch(ctx, escli, index, query) -} - -// helper to do a full-featured Lucene query parser (query_string) search, with all possible facets. Not safe to expose publicly. -func DoSearchGeneric(ctx context.Context, escli *es.Client, index, q string) (*EsSearchResponse, error) { - ctx, span := tracer.Start(ctx, "DoSearchGeneric") - defer span.End() - - query := map[string]any{ - "query": map[string]any{ - "query_string": map[string]any{ - "query": q, - "default_operator": "and", - "analyze_wildcard": true, - "allow_leading_wildcard": false, - "lenient": true, - "default_field": "everything", - }, - }, - } - - return doSearch(ctx, escli, index, query) -} - -func doSearch(ctx context.Context, escli *es.Client, index string, query any) (*EsSearchResponse, error) { - ctx, span := tracer.Start(ctx, "doSearch") - defer span.End() - - span.SetAttributes(attribute.String("index", index), attribute.String("query", fmt.Sprintf("%+v", query))) - - b, err := json.Marshal(query) - if err != nil { - return nil, fmt.Errorf("failed to serialize query: %w", err) - } - slog.Info("sending query", "index", index, "query", string(b)) - - // Perform the search request. - res, err := escli.Search( - escli.Search.WithContext(ctx), - escli.Search.WithIndex(index), - escli.Search.WithBody(bytes.NewBuffer(b)), - ) - if err != nil { - return nil, fmt.Errorf("search query error: %w", err) - } - defer res.Body.Close() - if res.IsError() { - raw, err := ioutil.ReadAll(res.Body) - if nil == err { - slog.Warn("search query error", "resp", string(raw), "status_code", res.StatusCode) - } - return nil, fmt.Errorf("search query error, code=%d", res.StatusCode) - } - - var out EsSearchResponse - if err := json.NewDecoder(res.Body).Decode(&out); err != nil { - return nil, fmt.Errorf("decoding search response: %w", err) - } - - return &out, nil -} diff --git a/search/query_test.go b/search/query_test.go deleted file mode 100644 index 50168780..00000000 --- a/search/query_test.go +++ /dev/null @@ -1,410 +0,0 @@ -//go:build localsearch - -package search - -import ( - "context" - "crypto/tls" - "io" - "log/slog" - "net/http" - "testing" - - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - - "github.com/ipfs/go-cid" - es "github.com/opensearch-project/opensearch-go/v2" - "github.com/stretchr/testify/assert" - "gorm.io/driver/sqlite" - "gorm.io/gorm" -) - -var ( - testPostIndex = "palomar_test_post" - testProfileIndex = "palomar_test_profile" -) - -func testEsClient(t *testing.T) *es.Client { - cfg := es.Config{ - Addresses: []string{"http://localhost:9200"}, - Username: "admin", - Password: "0penSearch-Pal0mar", - CACert: nil, - Transport: &http.Transport{ - MaxIdleConnsPerHost: 5, - TLSClientConfig: &tls.Config{ - InsecureSkipVerify: true, - }, - }, - } - escli, err := es.NewClient(cfg) - if err != nil { - t.Fatal(err) - } - info, err := escli.Info() - if err != nil { - t.Fatal(err) - } - info.Body.Close() - return escli - -} - -func testServer(ctx context.Context, t *testing.T, escli *es.Client, dir identity.Directory) *Server { - db, err := gorm.Open(sqlite.Open("file::memory:?cache=shared"), &gorm.Config{}) - if err != nil { - t.Fatal(err) - } - - srv, err := NewServer( - db, - escli, - dir, - Config{ - RelayHost: "wss://relay.invalid", - PostIndex: testPostIndex, - ProfileIndex: testProfileIndex, - Logger: slog.Default(), - RelaySyncRateLimit: 1, - IndexMaxConcurrency: 1, - }, - ) - if err != nil { - t.Fatal(err) - } - - // NOTE: skipping errors - resp, _ := srv.escli.Indices.Delete([]string{testPostIndex, testProfileIndex}) - defer resp.Body.Close() - io.ReadAll(resp.Body) - - if err := srv.EnsureIndices(ctx); err != nil { - t.Fatal(err) - } - - return srv -} - -func TestJapaneseRegressions(t *testing.T) { - assert := assert.New(t) - ctx := context.Background() - escli := testEsClient(t) - dir := identity.NewMockDirectory() - srv := testServer(ctx, t, escli, &dir) - ident := identity.Identity{ - DID: syntax.DID("did:plc:abc111"), - Handle: syntax.Handle("handle.example.com"), - } - - res, err := DoSearchPosts(ctx, &dir, escli, testPostIndex, "english", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(0, len(res.Hits.Hits)) - - p1 := appbsky.FeedPost{Text: "basic english post", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p1, "app.bsky.feed.post/3kpnillluoh2y", cid.Undef)) - - // https://github.com/bluesky-social/indigo/issues/302 - p2 := appbsky.FeedPost{Text: "学校から帰って熱いお風呂に入ったら力一杯がんばる", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p2, "app.bsky.feed.post/3kpnillluo222", cid.Undef)) - p3 := appbsky.FeedPost{Text: "熱力学", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p3, "app.bsky.feed.post/3kpnillluo333", cid.Undef)) - p4 := appbsky.FeedPost{Text: "東京都", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p4, "app.bsky.feed.post/3kpnillluo444", cid.Undef)) - p5 := appbsky.FeedPost{Text: "京都", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p5, "app.bsky.feed.post/3kpnillluo555", cid.Undef)) - p6 := appbsky.FeedPost{Text: "パリ", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p6, "app.bsky.feed.post/3kpnillluo666", cid.Undef)) - p7 := appbsky.FeedPost{Text: "ハリー・ポッター", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p7, "app.bsky.feed.post/3kpnillluo777", cid.Undef)) - p8 := appbsky.FeedPost{Text: "ハリ", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p8, "app.bsky.feed.post/3kpnillluo223", cid.Undef)) - p9 := appbsky.FeedPost{Text: "multilingual 多言語", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p9, "app.bsky.feed.post/3kpnillluo224", cid.Undef)) - - _, err = srv.escli.Indices.Refresh() - assert.NoError(err) - - // expect all to be indexed - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "*", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(9, len(res.Hits.Hits)) - - // check that english matches (single post) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "english", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // "thermodynamics"; should return only one match - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "熱力学", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // "Kyoto"; should return only one match - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "京都", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // "Paris"; should return only one match - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "パリ", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // should return only one match - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "ハリー", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // part of a word; should match none - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "ハ", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(0, len(res.Hits.Hits)) - - // should match both ways, and together - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "multilingual", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "多言語", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "multilingual 多言語", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "\"multilingual 多言語\"", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) -} - -func TestParsedQuery(t *testing.T) { - assert := assert.New(t) - ctx := context.Background() - escli := testEsClient(t) - dir := identity.NewMockDirectory() - srv := testServer(ctx, t, escli, &dir) - ident := identity.Identity{ - DID: syntax.DID("did:plc:abc111"), - Handle: syntax.Handle("handle.example.com"), - } - other := identity.Identity{ - DID: syntax.DID("did:plc:abc222"), - Handle: syntax.Handle("other.example.com"), - } - dir.Insert(ident) - dir.Insert(other) - - res, err := DoSearchPosts(ctx, &dir, escli, testPostIndex, "english", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(0, len(res.Hits.Hits)) - - p1 := appbsky.FeedPost{Text: "basic english post", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p1, "app.bsky.feed.post/3kpnillluoh2y", cid.Undef)) - p2 := appbsky.FeedPost{Text: "another english post", CreatedAt: "2024-01-02T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p2, "app.bsky.feed.post/3kpnilllu2222", cid.Undef)) - p3 := appbsky.FeedPost{ - Text: "#cat post with hashtag", - CreatedAt: "2024-01-02T03:04:05.006Z", - Facets: []*appbsky.RichtextFacet{ - &appbsky.RichtextFacet{ - Features: []*appbsky.RichtextFacet_Features_Elem{ - &appbsky.RichtextFacet_Features_Elem{ - RichtextFacet_Tag: &appbsky.RichtextFacet_Tag{ - Tag: "trick", - }, - }, - }, - Index: &appbsky.RichtextFacet_ByteSlice{ - ByteStart: 0, - ByteEnd: 4, - }, - }, - }, - } - assert.NoError(srv.indexPost(ctx, &ident, &p3, "app.bsky.feed.post/3kpnilllu3333", cid.Undef)) - p4 := appbsky.FeedPost{ - Text: "@other.example.com post with mention", - CreatedAt: "2024-01-02T03:04:05.006Z", - Facets: []*appbsky.RichtextFacet{ - &appbsky.RichtextFacet{ - Features: []*appbsky.RichtextFacet_Features_Elem{ - &appbsky.RichtextFacet_Features_Elem{ - RichtextFacet_Mention: &appbsky.RichtextFacet_Mention{ - Did: "did:plc:abc222", - }, - }, - }, - Index: &appbsky.RichtextFacet_ByteSlice{ - ByteStart: 0, - ByteEnd: 18, - }, - }, - }, - } - assert.NoError(srv.indexPost(ctx, &ident, &p4, "app.bsky.feed.post/3kpnilllu4444", cid.Undef)) - p5 := appbsky.FeedPost{ - Text: "https://bsky.app... post with hashtag #cat", - CreatedAt: "2024-01-02T03:04:05.006Z", - Facets: []*appbsky.RichtextFacet{ - &appbsky.RichtextFacet{ - Features: []*appbsky.RichtextFacet_Features_Elem{ - &appbsky.RichtextFacet_Features_Elem{ - RichtextFacet_Link: &appbsky.RichtextFacet_Link{ - Uri: "htTPS://www.en.wikipedia.org/wiki/CBOR?q=3&a=1&utm_campaign=123", - }, - }, - }, - Index: &appbsky.RichtextFacet_ByteSlice{ - ByteStart: 0, - ByteEnd: 19, - }, - }, - }, - } - assert.NoError(srv.indexPost(ctx, &ident, &p5, "app.bsky.feed.post/3kpnilllu5555", cid.Undef)) - p6 := appbsky.FeedPost{ - Text: "post with lang (deutsch)", - CreatedAt: "2024-01-02T03:04:05.006Z", - Langs: []string{"ja", "de-DE"}, - } - assert.NoError(srv.indexPost(ctx, &ident, &p6, "app.bsky.feed.post/3kpnilllu6666", cid.Undef)) - p7 := appbsky.FeedPost{Text: "post with old date", CreatedAt: "2020-05-03T03:04:05.006Z"} - assert.NoError(srv.indexPost(ctx, &ident, &p7, "app.bsky.feed.post/3kpnilllu7777", cid.Undef)) - - _, err = srv.escli.Indices.Refresh() - assert.NoError(err) - - // expect all to be indexed - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "*", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(7, len(res.Hits.Hits)) - - // check that english matches both - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "english", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(2, len(res.Hits.Hits)) - - // phrase only matches one - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "\"basic english\"", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // posts-by - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "from:handle.example.com", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(7, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "from:@handle.example.com", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(7, len(res.Hits.Hits)) - - // hashtag query - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "post #trick", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "post #Trick", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "post #trick #allMustMatch", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(0, len(res.Hits.Hits)) - - // mention query - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "@other.example.com", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // URL and domain queries - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "https://en.wikipedia.org/wiki/CBOR?a=1&q=3", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "\"https://en.wikipedia.org/wiki/CBOR?a=1&q=3\"", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(0, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "https://en.wikipedia.org/wiki/CBOR", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(0, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "domain:en.wikipedia.org", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // lang filter - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "lang:de", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - - // date range filters - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "since:2023-01-01T00:00:00Z", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(6, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "since:2023-01-01", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(6, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "until:2023-01-01", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(1, len(res.Hits.Hits)) - res, err = DoSearchPosts(ctx, &dir, escli, testPostIndex, "until:asdf", 0, 20) - if err != nil { - t.Fatal(err) - } - assert.Equal(7, len(res.Hits.Hits)) -} diff --git a/search/server.go b/search/server.go deleted file mode 100644 index 9cd9c92e..00000000 --- a/search/server.go +++ /dev/null @@ -1,168 +0,0 @@ -package search - -import ( - "context" - "fmt" - "io" - "log/slog" - "net/http" - "os" - "strings" - - _ "net/http/pprof" // For pprof in the metrics server - - "github.com/bluesky-social/indigo/atproto/identity" - - "github.com/earthboundkid/versioninfo/v2" - "github.com/labstack/echo/v4" - "github.com/labstack/echo/v4/middleware" - es "github.com/opensearch-project/opensearch-go/v2" - "github.com/prometheus/client_golang/prometheus/promhttp" - slogecho "github.com/samber/slog-echo" - "go.opentelemetry.io/contrib/instrumentation/github.com/labstack/echo/otelecho" -) - -type LastSeq struct { - ID uint `gorm:"primarykey"` - Seq int64 -} - -type ServerConfig struct { - Logger *slog.Logger - ProfileIndex string - PostIndex string - AtlantisAddresses []string -} - -type Server struct { - escli *es.Client - postIndex string - profileIndex string - dir identity.Directory - echo *echo.Echo - logger *slog.Logger - - Indexer *Indexer -} - -func NewServer(escli *es.Client, dir identity.Directory, config ServerConfig) (*Server, error) { - logger := config.Logger - if logger == nil { - logger = slog.New(slog.NewJSONHandler(os.Stdout, &slog.HandlerOptions{ - Level: slog.LevelInfo, - })) - } - - serv := Server{ - escli: escli, - postIndex: config.PostIndex, - profileIndex: config.ProfileIndex, - dir: dir, - logger: logger, - } - - return &serv, nil -} - -func (s *Server) EnsureIndices(ctx context.Context) error { - - indices := []struct { - Name string - SchemaJSON string - }{ - {Name: s.postIndex, SchemaJSON: palomarPostSchemaJSON}, - {Name: s.profileIndex, SchemaJSON: palomarProfileSchemaJSON}, - } - for _, idx := range indices { - resp, err := s.escli.Indices.Exists([]string{idx.Name}) - if err != nil { - return err - } - defer resp.Body.Close() - io.ReadAll(resp.Body) - if resp.IsError() && resp.StatusCode != 404 { - return fmt.Errorf("failed to check index existence") - } - if resp.StatusCode == 404 { - s.logger.Warn("creating opensearch index", "index", idx.Name) - if len(idx.SchemaJSON) < 2 { - return fmt.Errorf("empty schema file (go:embed failed)") - } - buf := strings.NewReader(idx.SchemaJSON) - resp, err := s.escli.Indices.Create( - idx.Name, - s.escli.Indices.Create.WithBody(buf)) - if err != nil { - return err - } - defer resp.Body.Close() - errBytes, err := io.ReadAll(resp.Body) - if resp.IsError() { - s.logger.Error("failed to create index", "index", idx.Name, "response", string(errBytes)) - return fmt.Errorf("failed to create index") - } - if err != nil { - return err - } - } - } - return nil -} - -type HealthStatus struct { - Service string `json:"service,const=palomar"` - Status string `json:"status"` - Version string `json:"version"` - Message string `json:"msg,omitempty"` -} - -func (a *Server) handleHealthCheck(c echo.Context) error { - if a.Indexer != nil { - if err := a.Indexer.db.Exec("SELECT 1").Error; err != nil { - a.logger.Error("healthcheck can't connect to database", "err", err) - return c.JSON(500, HealthStatus{Status: "error", Version: versioninfo.Short(), Message: "can't connect to database"}) - } - } - return c.JSON(200, HealthStatus{Status: "ok", Version: versioninfo.Short()}) -} - -func (s *Server) RunAPI(listen string) error { - - s.logger.Info("Configuring HTTP server") - e := echo.New() - e.HideBanner = true - e.Use(slogecho.New(s.logger)) - e.Use(middleware.Recover()) - e.Use(MetricsMiddleware) - e.Use(middleware.BodyLimit("64M")) - e.Use(otelecho.Middleware("palomar")) - - e.HTTPErrorHandler = func(err error, ctx echo.Context) { - code := 500 - if he, ok := err.(*echo.HTTPError); ok { - code = he.Code - } - s.logger.Warn("HTTP request error", "statusCode", code, "path", ctx.Path(), "err", err) - ctx.Response().WriteHeader(code) - } - - e.Use(middleware.CORS()) - e.GET("/", s.handleHealthCheck) - e.GET("/_health", s.handleHealthCheck) - e.GET("/metrics", echo.WrapHandler(promhttp.Handler())) - e.GET("/xrpc/app.bsky.unspecced.searchPostsSkeleton", s.handleSearchPostsSkeleton) - e.GET("/xrpc/app.bsky.unspecced.searchActorsSkeleton", s.handleSearchActorsSkeleton) - s.echo = e - - s.logger.Info("starting search API daemon", "bind", listen) - return s.echo.Start(listen) -} - -func (s *Server) RunMetrics(listen string) error { - http.Handle("/metrics", promhttp.Handler()) - return http.ListenAndServe(listen, nil) -} - -func (s *Server) Shutdown(ctx context.Context) error { - return s.echo.Shutdown(ctx) -} diff --git a/search/testdata/transform-post-fixtures.json b/search/testdata/transform-post-fixtures.json deleted file mode 100644 index 780a9705..00000000 --- a/search/testdata/transform-post-fixtures.json +++ /dev/null @@ -1,344 +0,0 @@ -[ - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "rkey": "3k4duaz5vfs2b", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "PostRecord": { - "$type": "app.bsky.feed.post", - "text": "post which embeds an external URL as a card", - "createdAt": "2023-08-07T05:46:14.423045Z", - "embed": { - "$type": "app.bsky.embed.external", - "external": { - "uri": "https://www.bsky.app:443/index.html", - "title": "Bluesky Social", - "description": "See what's next.", - "thumb": { - "$type": "blob", - "ref": { - "$link": "bafkreiash5eihfku2jg4skhyh5kes7j5d5fd6xxloaytdywcvb3r3zrzhu" - }, - "mimeType": "image/png", - "size": 23527 - } - } - } - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn_3k4duaz5vfs2b", - "PostDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_rkey": "3k4duaz5vfs2b", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "created_at": "2023-08-07T05:46:14.423045Z", - "text": "post which embeds an external URL as a card", - "url": [ - "https://bsky.app" - ], - "domain": [ - "bsky.app" - ], - "embed_img_count": 0 - } - }, - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "rkey": "3k4duaz5vfs2b", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "PostRecord": { - "$type": "app.bsky.feed.post", - "text": "longer example with #some #hashtags, emoji \u2620 \ud83d\ude42 \ud83c\udf85\ud83c\udfff, flags \ud83c\uddf8\ud83c\udde8 ", - "createdAt": "2023-08-07T05:46:14.423045Z", - "langs": [ - "th", - "en-US" - ], - "facets": [ - { - "index": { - "byteStart": 23, - "byteEnd": 35 - }, - "features": [ - { - "$type": "app.bsky.richtext.facet#mention", - "did": "did:plc:ewvi7nxzyoun6zhxrhs64oiz" - } - ] - }, - { - "index": { - "byteStart": 74, - "byteEnd": 108 - }, - "features": [ - { - "$type": "app.bsky.richtext.facet#link", - "uri": "https://en.wikipedia.org/wiki/CBOR?utm_campaign=123" - } - ] - }, - { - "index": { - "byteStart": 21, - "byteEnd": 25 - }, - "features": [ - { - "$type": "app.bsky.richtext.facet#tag", - "tag": "some" - } - ] - } - ], - "labels": { - "$type": "com.atproto.label.defs#selfLabels", - "values": [ - { - "val": "nudity" - } - ] - }, - "reply": { - "root": { - "uri": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k43tv4rft22g", - "cid": "bafyreig2fjxi3rptqdgylg7e5hmjl6mcke7rn2b6cugzlqq3i4zu6rq52q" - }, - "parent": { - "uri": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k43tv4rft22g", - "cid": "bafyreig2fjxi3rptqdgylg7e5hmjl6mcke7rn2b6cugzlqq3i4zu6rq52q" - } - }, - "embed": { - "$type": "app.bsky.embed.record", - "record": { - "uri": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k44deefqdk2g", - "cid": "bafyreiecx6dujwoeqpdzl27w67z4h46hyklk3an4i4cvvmioaqb2qbyo5u" - } - }, - "tags": [ - "thing" - ] - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn_3k4duaz5vfs2b", - "PostDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_rkey": "3k4duaz5vfs2b", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "created_at": "2023-08-07T05:46:14.423045Z", - "text": "longer example with #some #hashtags, emoji \u2620 \ud83d\ude42 \ud83c\udf85\ud83c\udfff, flags \ud83c\uddf8\ud83c\udde8 ", - "reply_root_aturi": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k43tv4rft22g", - "mention_did": [ - "did:plc:ewvi7nxzyoun6zhxrhs64oiz" - ], - "embed_aturi": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k44deefqdk2g", - "lang_code": [ - "th", - "en-US" - ], - "lang_code_iso2": [ - "th", - "en" - ], - "self_label": [ - "nudity" - ], - "url": [ - "https://en.wikipedia.org/wiki/CBOR" - ], - "domain": [ - "en.wikipedia.org" - ], - "tag": [ - "some", - "thing" - ], - "emoji": [ - "\u2620", - "\ud83d\ude42", - "\ud83c\udf85\ud83c\udfff", - "\ud83c\uddf8\ud83c\udde8" - ], - "embed_img_count": 0 - } - }, - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "rkey": "3k4duaz5vfs2b", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "PostRecord": { - "$type": "app.bsky.feed.post", - "text": "", - "createdAt": "2023-08-07T05:46:14.423045Z", - "embed": { - "$type": "app.bsky.embed.images", - "images": [ - { - "alt": "brief alt text description of the first image", - "image": { - "$type": "blob", - "ref": { - "$link": "bafkreibabalobzn6cd366ukcsjycp4yymjymgfxcv6xczmlgpemzkz3cfa" - }, - "mimeType": "image/webp", - "size": 760898 - } - }, - { - "alt": "brief alt text description of the second image", - "image": { - "$type": "blob", - "ref": { - "$link": "bafkreif3fouono2i3fmm5moqypwskh3yjtp7snd5hfq5pr453oggygyrte" - }, - "mimeType": "image/png", - "size": 13208 - } - } - ] - } - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn_3k4duaz5vfs2b", - "PostDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_rkey": "3k4duaz5vfs2b", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "created_at": "2023-08-07T05:46:14.423045Z", - "text": "", - "embed_img_alt_text": [ - "brief alt text description of the first image", - "brief alt text description of the second image" - ], - "embed_img_count": 2 - } - }, - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "rkey": "3k4duaz5vfs2b", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "PostRecord": { - "$type": "app.bsky.feed.post", - "text": "学校から帰って熱いお風呂に入ったら力一杯がんばる", - "createdAt": "2023-08-07T05:46:14.423045Z", - "embed": { - "$type": "app.bsky.embed.images", - "images": [ - { - "alt": "brief alt text description of the first image ハリー・ポッター", - "image": { - "$type": "blob", - "ref": { - "$link": "bafkreibabalobzn6cd366ukcsjycp4yymjymgfxcv6xczmlgpemzkz3cfa" - }, - "mimeType": "image/webp", - "size": 760898 - } - }, - { - "alt": "brief alt text description of the second image", - "image": { - "$type": "blob", - "ref": { - "$link": "bafkreif3fouono2i3fmm5moqypwskh3yjtp7snd5hfq5pr453oggygyrte" - }, - "mimeType": "image/png", - "size": 13208 - } - } - ] - } - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn_3k4duaz5vfs2b", - "PostDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_rkey": "3k4duaz5vfs2b", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "created_at": "2023-08-07T05:46:14.423045Z", - "text": "学校から帰って熱いお風呂に入ったら力一杯がんばる", - "text_ja": "学校から帰って熱いお風呂に入ったら力一杯がんばる", - "embed_img_alt_text": [ - "brief alt text description of the first image ハリー・ポッター", - "brief alt text description of the second image" - ], - "embed_img_alt_text_ja": [ - "brief alt text description of the first image ハリー・ポッター" - ], - "embed_img_count": 2 - } - }, - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "rkey": "3k4duaz5vfs2d", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "PostRecord": { - "$type": "app.bsky.feed.post", - "text": "", - "createdAt": "2023-08-07T05:46:14.423045Z", - "embed": { - "$type": "app.bsky.embed.recordWithMedia", - "media": { - "$type": "app.bsky.embed.images", - "images": [ - { - "alt": "brief alt text description of the first image", - "image": { - "$type": "blob", - "ref": { - "$link": "bafkreibabalobzn6cd366ukcsjycp4yymjymgfxcv6xczmlgpemzkz3cfa" - }, - "mimeType": "image/webp", - "size": 760898 - } - }, - { - "alt": "brief alt text description of the second image", - "image": { - "$type": "blob", - "ref": { - "$link": "bafkreif3fouono2i3fmm5moqypwskh3yjtp7snd5hfq5pr453oggygyrte" - }, - "mimeType": "image/png", - "size": 13208 - } - } - ] - }, - "record": { - "$type": "app.bsky.embed.record", - "record": { - "cid": "bafyreiecx6dujwoeqpdzl27w67z4h46hyklk3an4i4cvvmioaqb2qbyo5u", - "uri": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k44deefqdk2g" - } - } - } - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn_3k4duaz5vfs2d", - "PostDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_rkey": "3k4duaz5vfs2d", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "created_at": "2023-08-07T05:46:14.423045Z", - "text": "", - "embed_img_alt_text": [ - "brief alt text description of the first image", - "brief alt text description of the second image" - ], - "embed_img_count": 2, - "embed_aturi": "at://did:plc:u5cwb2mwiv2bfq53cjufe6yn/app.bsky.feed.post/3k44deefqdk2g" - } - } -] diff --git a/search/testdata/transform-profile-fixtures.json b/search/testdata/transform-profile-fixtures.json deleted file mode 100644 index 4b2f5282..00000000 --- a/search/testdata/transform-profile-fixtures.json +++ /dev/null @@ -1,62 +0,0 @@ -[ - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "rkey": "self", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "handle": "handle.example.com", - "ProfileRecord": { - "$type": "app.bsky.actor.profile" - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "ProfileDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "has_avatar": false, - "has_banner": false - } - }, - { - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "rkey": "self", - "cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "handle": "handle.example.com", - "ProfileRecord": { - "$type": "app.bsky.actor.profile", - "displayName": "Big Bubba", - "description": "Big Description 🥸 #cheese", - "labels": { - "$type": "com.atproto.label.defs#selfLabels", - "values": [ - {"val": "nudity"} - ] - }, - "avatar": { - "$type": "blob", - "mimeType": "image/jpeg", - "ref": { - "$link": "bafkreiglnysron3h2je7nf6cmvtimuaxi7xe2c7rkxitmks3mzmajnc2ou" - }, - "size": 106760 - }, - "banner": { - "cid": "bafkreih42ufe7ies4zephtkdw4tcb4hvuoc2pjzybpjxpecz3tyymnvkhm", - "mimeType": "image/jpeg" - } - }, - "doc_id": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "ProfileDoc": { - "doc_index_ts": "2006-01-02T15:04:05.000Z", - "did": "did:plc:u5cwb2mwiv2bfq53cjufe6yn", - "handle": "handle.example.com", - "record_cid": "bafyreibjifzpqj6o6wcq3hejh7y4z4z2vmiklkvykc57tw3pcbx3kxifpm", - "display_name": "Big Bubba", - "description": "Big Description 🥸 #cheese", - "self_label": ["nudity"], - "emoji": ["🥸"], - "has_avatar": true, - "has_banner": true - } - } -] diff --git a/search/transform.go b/search/transform.go deleted file mode 100644 index 917b622d..00000000 --- a/search/transform.go +++ /dev/null @@ -1,296 +0,0 @@ -package search - -import ( - "log/slog" - "net/url" - "strings" - "time" - - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - - "github.com/rivo/uniseg" -) - -type ProfileDoc struct { - DocIndexTs string `json:"doc_index_ts"` - DID string `json:"did"` - RecordCID string `json:"record_cid"` - Handle string `json:"handle"` - DisplayName *string `json:"display_name,omitempty"` - Description *string `json:"description,omitempty"` - ImgAltText []string `json:"img_alt_text,omitempty"` - SelfLabel []string `json:"self_label,omitempty"` - URL []string `json:"url,omitempty"` - Domain []string `json:"domain,omitempty"` - Tag []string `json:"tag,omitempty"` - Emoji []string `json:"emoji,omitempty"` - HasAvatar bool `json:"has_avatar"` - HasBanner bool `json:"has_banner"` -} - -type PostDoc struct { - DocIndexTs string `json:"doc_index_ts"` - DID string `json:"did"` - RecordRkey string `json:"record_rkey"` - RecordCID string `json:"record_cid"` - CreatedAt *string `json:"created_at,omitempty"` - Text string `json:"text"` - TextJA *string `json:"text_ja,omitempty"` - LangCode []string `json:"lang_code,omitempty"` - LangCodeIso2 []string `json:"lang_code_iso2,omitempty"` - MentionDID []string `json:"mention_did,omitempty"` - EmbedATURI *string `json:"embed_aturi,omitempty"` - ReplyRootATURI *string `json:"reply_root_aturi,omitempty"` - EmbedImgCount int `json:"embed_img_count"` - EmbedImgAltText []string `json:"embed_img_alt_text,omitempty"` - EmbedImgAltTextJA []string `json:"embed_img_alt_text_ja,omitempty"` - SelfLabel []string `json:"self_label,omitempty"` - URL []string `json:"url,omitempty"` - Domain []string `json:"domain,omitempty"` - Tag []string `json:"tag,omitempty"` - Emoji []string `json:"emoji,omitempty"` -} - -// Returns the search index document ID (`_id`) for this document. -// -// This identifier should be URL safe and not contain a slash ("/"). -func (d *ProfileDoc) DocId() string { - return d.DID -} - -// Returns the search index document ID (`_id`) for this document. -// -// This identifier should be URL safe and not contain a slash ("/"). -func (d *PostDoc) DocId() string { - return d.DID + "_" + d.RecordRkey -} - -func TransformProfile(profile *appbsky.ActorProfile, ident *identity.Identity, cid string) ProfileDoc { - // TODO: placeholder for future alt text on profile blobs - var altText []string - var tags []string - var emojis []string - if profile.Description != nil { - tags = parseProfileTags(profile) - emojis = parseEmojis(*profile.Description) - } - var selfLabels []string - if profile.Labels != nil && profile.Labels.LabelDefs_SelfLabels != nil { - for _, le := range profile.Labels.LabelDefs_SelfLabels.Values { - selfLabels = append(selfLabels, le.Val) - } - } - handle := "" - if !ident.Handle.IsInvalidHandle() { - handle = ident.Handle.String() - } - return ProfileDoc{ - DocIndexTs: syntax.DatetimeNow().String(), - DID: ident.DID.String(), - RecordCID: cid, - Handle: handle, - DisplayName: profile.DisplayName, - Description: profile.Description, - ImgAltText: altText, - SelfLabel: selfLabels, - Tag: tags, - Emoji: emojis, - HasAvatar: profile.Avatar != nil, - HasBanner: profile.Banner != nil, - } -} - -func TransformPost(post *appbsky.FeedPost, did syntax.DID, rkey, cid string) PostDoc { - altText := []string{} - if post.Embed != nil && post.Embed.EmbedImages != nil { - for _, img := range post.Embed.EmbedImages.Images { - if img.Alt != "" { - altText = append(altText, img.Alt) - } - } - } - var langCodeIso2 []string - for _, lang := range post.Langs { - // TODO: include an actual language code map to go from 3char to 2char - prefix := strings.SplitN(lang, "-", 2)[0] - if len(prefix) == 2 { - langCodeIso2 = append(langCodeIso2, strings.ToLower(prefix)) - } - } - var mentionDIDs []string - var urls []string - for _, facet := range post.Facets { - for _, feat := range facet.Features { - if feat.RichtextFacet_Mention != nil { - mentionDIDs = append(mentionDIDs, feat.RichtextFacet_Mention.Did) - } - if feat.RichtextFacet_Link != nil { - urls = append(urls, feat.RichtextFacet_Link.Uri) - } - } - } - var replyRootATURI *string - if post.Reply != nil { - replyRootATURI = &(post.Reply.Root.Uri) - } - if post.Embed != nil && post.Embed.EmbedExternal != nil { - urls = append(urls, post.Embed.EmbedExternal.External.Uri) - } - var embedATURI *string - if post.Embed != nil && post.Embed.EmbedRecord != nil { - embedATURI = &post.Embed.EmbedRecord.Record.Uri - } - if post.Embed != nil && post.Embed.EmbedRecordWithMedia != nil { - embedATURI = &post.Embed.EmbedRecordWithMedia.Record.Record.Uri - } - var embedImgCount int - var embedImgAltText []string - var embedImgAltTextJA []string - if post.Embed != nil && post.Embed.EmbedImages != nil { - embedImgCount = len(post.Embed.EmbedImages.Images) - for _, img := range post.Embed.EmbedImages.Images { - if img.Alt != "" { - embedImgAltText = append(embedImgAltText, img.Alt) - if containsJapanese(img.Alt) { - embedImgAltTextJA = append(embedImgAltTextJA, img.Alt) - } - } - } - } - - if post.Embed != nil && - post.Embed.EmbedRecordWithMedia != nil && - post.Embed.EmbedRecordWithMedia.Media != nil && - post.Embed.EmbedRecordWithMedia.Media.EmbedImages != nil && - len(post.Embed.EmbedRecordWithMedia.Media.EmbedImages.Images) > 0 { - embedImgCount += len(post.Embed.EmbedRecordWithMedia.Media.EmbedImages.Images) - for _, img := range post.Embed.EmbedRecordWithMedia.Media.EmbedImages.Images { - if img.Alt != "" { - embedImgAltText = append(embedImgAltText, img.Alt) - if containsJapanese(img.Alt) { - embedImgAltTextJA = append(embedImgAltTextJA, img.Alt) - } - } - } - } - - var selfLabels []string - if post.Labels != nil && post.Labels.LabelDefs_SelfLabels != nil { - for _, le := range post.Labels.LabelDefs_SelfLabels.Values { - selfLabels = append(selfLabels, le.Val) - } - } - - var domains []string - for i, raw := range urls { - clean := NormalizeLossyURL(raw) - urls[i] = clean - u, err := url.Parse(clean) - if nil == err { - domains = append(domains, u.Hostname()) - } - } - - doc := PostDoc{ - DocIndexTs: syntax.DatetimeNow().String(), - DID: did.String(), - RecordRkey: rkey, - RecordCID: cid, - Text: post.Text, - LangCode: post.Langs, - LangCodeIso2: langCodeIso2, - MentionDID: mentionDIDs, - EmbedATURI: embedATURI, - ReplyRootATURI: replyRootATURI, - EmbedImgCount: embedImgCount, - EmbedImgAltText: embedImgAltText, - EmbedImgAltTextJA: embedImgAltTextJA, - SelfLabel: selfLabels, - URL: urls, - Domain: domains, - Tag: parsePostTags(post), - Emoji: parseEmojis(post.Text), - } - - if containsJapanese(post.Text) { - doc.TextJA = &post.Text - } - - if post.CreatedAt != "" { - // there are some old bad timestamps out there! - dt, err := syntax.ParseDatetimeLenient(post.CreatedAt) - if nil == err { // *not* an error - // not more than a few minutes in the future - if time.Since(dt.Time()) >= -1*5*time.Minute { - s := dt.String() - doc.CreatedAt = &s - } else { - slog.Warn("rejecting future post CreatedAt", "datetime", dt.String(), "did", did.String(), "rkey", rkey) - s := syntax.DatetimeNow().String() - doc.CreatedAt = &s - } - } - } - - return doc -} - -func dedupeStrings(in []string) []string { - var out []string - seen := make(map[string]bool) - for _, v := range in { - if !seen[v] { - out = append(out, v) - seen[v] = true - } - } - return out -} - -func parseProfileTags(p *appbsky.ActorProfile) []string { - // TODO: waiting for profile tag lexicon support - var ret []string = []string{} - if len(ret) == 0 { - return nil - } - return dedupeStrings(ret) -} - -func parsePostTags(p *appbsky.FeedPost) []string { - var ret []string = []string{} - for _, facet := range p.Facets { - for _, feat := range facet.Features { - if feat.RichtextFacet_Tag != nil { - ret = append(ret, feat.RichtextFacet_Tag.Tag) - } - } - } - ret = append(ret, p.Tags...) - if len(ret) == 0 { - return nil - } - return dedupeStrings(ret) -} - -func parseEmojis(s string) []string { - var ret []string = []string{} - seen := make(map[string]bool) - gr := uniseg.NewGraphemes(s) - for gr.Next() { - // check if this grapheme cluster starts with an emoji rune (Unicode codepoint, int32) - firstRune := gr.Runes()[0] - if (firstRune >= 0x1F000 && firstRune <= 0x1FFFF) || (firstRune >= 0x2600 && firstRune <= 0x26FF) { - emoji := gr.Str() - if seen[emoji] == false { - ret = append(ret, emoji) - seen[emoji] = true - } - } - } - if len(ret) == 0 { - return nil - } - return ret -} diff --git a/search/transform_test.go b/search/transform_test.go deleted file mode 100644 index c91542c1..00000000 --- a/search/transform_test.go +++ /dev/null @@ -1,115 +0,0 @@ -package search - -import ( - "encoding/json" - "io" - "os" - "testing" - - appbsky "github.com/bluesky-social/indigo/api/bsky" - "github.com/bluesky-social/indigo/atproto/identity" - "github.com/bluesky-social/indigo/atproto/syntax" - - "github.com/stretchr/testify/assert" -) - -func TestParseEmojis(t *testing.T) { - assert := assert.New(t) - - assert.Equal(parseEmojis("bunch 🎅 of 🏡 emoji 🤰and 🫄 some 👩‍👩‍👧‍👧 compound"), []string{"🎅", "🏡", "🤰", "🫄", "👩‍👩‍👧‍👧"}) - - assert.Equal(parseEmojis("more ⛄ from ☠ lower ⛴ range"), []string{"⛄", "☠", "⛴"}) - assert.True(parseEmojis("blah") == nil) -} - -type profileFixture struct { - DID string `json:"did"` - Handle string `json:"handle"` - Rkey string `json:"rkey"` - Cid string `json:"cid"` - DocId string `json:"doc_id"` - ProfileRecord *appbsky.ActorProfile - ProfileDoc ProfileDoc -} - -func TestTransformProfileFixtures(t *testing.T) { - f, err := os.Open("testdata/transform-profile-fixtures.json") - if err != nil { - t.Fatal(err) - } - defer f.Close() - - fixBytes, err := io.ReadAll(f) - if err != nil { - t.Fatal(err) - } - - var fixtures []profileFixture - if err := json.Unmarshal(fixBytes, &fixtures); err != nil { - t.Fatal(err) - } - - for _, row := range fixtures { - _ = row - testProfileFixture(t, row) - } -} - -func testProfileFixture(t *testing.T, row profileFixture) { - assert := assert.New(t) - - repo := identity.Identity{ - Handle: syntax.Handle(row.Handle), - DID: syntax.DID(row.DID), - } - doc := TransformProfile(row.ProfileRecord, &repo, row.Cid) - doc.DocIndexTs = "2006-01-02T15:04:05.000Z" - assert.Equal(row.ProfileDoc, doc) - assert.Equal(row.DocId, doc.DocId()) -} - -type postFixture struct { - DID string `json:"did"` - Handle string `json:"handle"` - Rkey string `json:"rkey"` - Cid string `json:"cid"` - DocId string `json:"doc_id"` - PostRecord *appbsky.FeedPost - PostDoc PostDoc -} - -func TestTransformPostFixtures(t *testing.T) { - f, err := os.Open("testdata/transform-post-fixtures.json") - if err != nil { - t.Fatal(err) - } - defer f.Close() - - fixBytes, err := io.ReadAll(f) - if err != nil { - t.Fatal(err) - } - - var fixtures []postFixture - if err := json.Unmarshal(fixBytes, &fixtures); err != nil { - t.Fatal(err) - } - - for _, row := range fixtures { - _ = row - testPostFixture(t, row) - } -} - -func testPostFixture(t *testing.T, row postFixture) { - assert := assert.New(t) - - repo := identity.Identity{ - Handle: syntax.Handle(row.Handle), - DID: syntax.DID(row.DID), - } - doc := TransformPost(row.PostRecord, repo.DID, row.Rkey, row.Cid) - doc.DocIndexTs = "2006-01-02T15:04:05.000Z" - assert.Equal(row.PostDoc, doc) - assert.Equal(row.DocId, doc.DocId()) -} diff --git a/search/url.go b/search/url.go deleted file mode 100644 index 60678051..00000000 --- a/search/url.go +++ /dev/null @@ -1,61 +0,0 @@ -package search - -import ( - "net/url" - - "github.com/PuerkitoBio/purell" -) - -var trackingParams = []string{ - "__s", - "_ga", - "campaign_id", - "ceid", - "emci", - "emdi", - "fbclid", - "gclid", - "hootPostID", - "mc_eid", - "mkclid", - "mkt_tok", - "msclkid", - "pk_campaign", - "pk_kwd", - "sessionid", - "sourceid", - "utm_campaign", - "utm_content", - "utm_id", - "utm_medium", - "utm_source", - "utm_term", - "xpid", -} - -// aggressively normalizes URL, for search indexing and matching. it is possible the URL won't be directly functional after this normalization -func NormalizeLossyURL(raw string) string { - clean, err := purell.NormalizeURLString(raw, purell.FlagsUsuallySafeGreedy|purell.FlagRemoveDirectoryIndex|purell.FlagRemoveFragment|purell.FlagRemoveDuplicateSlashes|purell.FlagRemoveWWW|purell.FlagSortQuery) - if err != nil { - return raw - } - - // remove tracking params - u, err := url.Parse(clean) - if err != nil { - return clean - } - if u.RawQuery == "" { - return clean - } - params := u.Query() - - // there is probably a more efficient way to do this - for _, p := range trackingParams { - if params.Has(p) { - params.Del(p) - } - } - u.RawQuery = params.Encode() - return u.String() -} diff --git a/search/url_test.go b/search/url_test.go deleted file mode 100644 index 663bbc06..00000000 --- a/search/url_test.go +++ /dev/null @@ -1,33 +0,0 @@ -package search - -import ( - "testing" - - "github.com/stretchr/testify/assert" -) - -func TestNormalizeLossyURL(t *testing.T) { - assert := assert.New(t) - - fixtures := []struct { - orig string - clean string - }{ - {orig: "", clean: ""}, - {orig: "asdf", clean: "asdf"}, - {orig: "HTTP://bSky.app:80/index.html", clean: "http://bsky.app"}, - {orig: "https://example.com/thing?c=123&utm_campaign=blah&a=first", clean: "https://example.com/thing?a=first&c=123"}, - {orig: "https://example.com/thing?c=123&utm_campaign=blah&a=first", clean: "https://example.com/thing?a=first&c=123"}, - {orig: "http://example.com/foo//bar.html", clean: "http://example.com/foo/bar.html"}, - {orig: "http://example.com/bar.html#section1", clean: "http://example.com/bar.html"}, - {orig: "http://example.com/foo/", clean: "http://example.com/foo"}, - {orig: "http://example.com/", clean: "http://example.com"}, - {orig: "http://example.com/%7Efoo", clean: "http://example.com/~foo"}, - {orig: "http://example.com/foo/./bar/baz/../qux", clean: "http://example.com/foo/bar/qux"}, - {orig: "http://www.example.com/", clean: "http://example.com"}, - } - - for _, fix := range fixtures { - assert.Equal(fix.clean, NormalizeLossyURL(fix.orig)) - } -}