Skip to content

Commit e9a9240

Browse files
authored
feat: add v3 search endpoint and deprecate v1 search (#2688)
* surrounding * write tokens * implementation * review comments * address comments * include microblock canonical
1 parent 69e0f2f commit e9a9240

17 files changed

Lines changed: 2185 additions & 4 deletions

File tree

‎.github/workflows/ci.yml‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -67,6 +67,7 @@ jobs:
6767
api:principal,
6868
api:principal-v3,
6969
api:redis,
70+
api:search,
7071
api:smart-contracts,
7172
api:sockets,
7273
api:synthetic-txs,
Lines changed: 86 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,86 @@
1+
import type { ColumnDefinitions, MigrationBuilder } from 'node-pg-migrate';
2+
3+
export const shorthands: ColumnDefinitions | undefined = undefined;
4+
5+
/**
6+
* Indexes backing the name matching in the v3 search endpoint.
7+
*
8+
* Search resolves a partial term against contract ids and principals, which the existing indexes
9+
* cannot serve: the default btrees use the database collation, so they are unusable for `LIKE
10+
* 'term%'` prefix scans, and no index at all can serve the substring match (`ILIKE '%term%'`) that
11+
* powers fuzzy contract-name search. Two index kinds cover both:
12+
*
13+
* - `text_pattern_ops` btrees for prefix matching, on `smart_contracts.contract_id` (a partial
14+
* contract principal such as `SP2C2….arkadi`) and `principal_tx_counts.principal` (a partial
15+
* account address). The latter table holds one row per principal ever seen (millions, versus the
16+
* billions of rows in `txs` and the event tables) so prefix search never touches the large
17+
* tables, and its index only gains an entry when a principal is seen for the first time.
18+
* - A `pg_trgm` GIN index on `smart_contracts.contract_id` for substring and similarity matching,
19+
* which also supplies the `similarity()` ranking score. `smart_contracts` gains one row per
20+
* contract deploy, so GIN maintenance is negligible on the ingestion path.
21+
*
22+
* `pg_trgm` is optional. It is a trusted extension (PG 13+), so the database owner can install it
23+
* without superuser, but an operator whose server does not ship it (or whose role lacks `CREATE` on
24+
* the database) must still be able to migrate. Both failures are caught here and the trigram index
25+
* is skipped; the API detects its absence at runtime and falls back to unindexed substring
26+
* matching, which is correct but slower and ranks by match position instead of similarity.
27+
*
28+
* Installing the extension after this migration has run does NOT create the skipped index:
29+
* node-pg-migrate records a migration once and never re-runs it. An operator who adds `pg_trgm`
30+
* later should create the indexes by hand:
31+
*
32+
* CREATE EXTENSION IF NOT EXISTS pg_trgm; CREATE INDEX CONCURRENTLY IF NOT EXISTS
33+
* smart_contracts_contract_id_trgm_idx ON smart_contracts USING gin (contract_id gin_trgm_ops);
34+
* CREATE INDEX CONCURRENTLY IF NOT EXISTS token_assets_asset_identifier_trgm_idx ON token_assets
35+
* USING gin (asset_identifier gin_trgm_ops);
36+
*/
37+
export function up(pgm: MigrationBuilder): void {
38+
// `CREATE EXTENSION` inside an exception block runs as a subtransaction, so a failure rolls back
39+
// to the savepoint and leaves this migration's transaction intact.
40+
pgm.sql(`
41+
DO $$
42+
BEGIN
43+
CREATE EXTENSION IF NOT EXISTS pg_trgm;
44+
EXCEPTION WHEN OTHERS THEN
45+
RAISE WARNING 'Could not install the pg_trgm extension (%). Fuzzy name matching in the v3 search endpoint will fall back to unindexed substring matching. Install pg_trgm and create the trigram indexes by hand to enable it.', SQLERRM;
46+
END
47+
$$;
48+
`);
49+
50+
pgm.createIndex('smart_contracts', [{ name: 'contract_id', opclass: 'text_pattern_ops' }], {
51+
name: 'smart_contracts_contract_id_pattern_idx',
52+
});
53+
pgm.createIndex('principal_tx_counts', [{ name: 'principal', opclass: 'text_pattern_ops' }], {
54+
name: 'principal_tx_counts_principal_pattern_idx',
55+
});
56+
57+
// Deferred through `EXECUTE` so `gin_trgm_ops` is never parsed when the extension is absent.
58+
pgm.sql(`
59+
DO $$
60+
BEGIN
61+
IF EXISTS (SELECT 1 FROM pg_extension WHERE extname = 'pg_trgm') THEN
62+
EXECUTE 'CREATE INDEX smart_contracts_contract_id_trgm_idx
63+
ON smart_contracts USING gin (contract_id gin_trgm_ops)';
64+
END IF;
65+
END
66+
$$;
67+
`);
68+
}
69+
70+
export function down(pgm: MigrationBuilder): void {
71+
pgm.dropIndex('principal_tx_counts', [], {
72+
name: 'principal_tx_counts_principal_pattern_idx',
73+
ifExists: true,
74+
});
75+
pgm.dropIndex('smart_contracts', [], {
76+
name: 'smart_contracts_contract_id_trgm_idx',
77+
ifExists: true,
78+
});
79+
pgm.dropIndex('smart_contracts', [], {
80+
name: 'smart_contracts_contract_id_pattern_idx',
81+
ifExists: true,
82+
});
83+
// The extension is deliberately left installed. `up` creates it with `IF NOT EXISTS`, so it may
84+
// well predate this migration and be shared with objects this migration knows nothing about;
85+
// dropping it could remove something another feature depends on, or fail outright.
86+
}
Lines changed: 155 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,155 @@
1+
import type { ColumnDefinitions, MigrationBuilder } from 'node-pg-migrate';
2+
3+
export const shorthands: ColumnDefinitions | undefined = undefined;
4+
5+
/**
6+
* Distinct on-chain token assets, so search can match a term against an asset identifier without
7+
* scanning the event tables.
8+
*
9+
* Asset identifiers are otherwise only recorded per occurrence: `ft_events`/`nft_events` hold one
10+
* row per transfer and `ft_balances` one row per address/token pair, all of them far too large to
11+
* substring-match or even to `DISTINCT` on demand. This table holds one row per asset identifier
12+
* ever seen (order of tens of thousands) which makes a trigram index over it trivially small.
13+
*
14+
* Only the on-chain identifier is stored (`SP….contract-name::asset-name`). The display name and
15+
* symbol a token reports through `get-name`/`get-symbol` are runtime contract-call results that
16+
* this API never reads; they belong to token-metadata-api.
17+
*
18+
* Rows are never removed on a re-org, because a re-org does not re-insert the events that would
19+
* recreate them: an asset first seen on a block that is later promoted would be lost forever.
20+
* Instead `tx_id` records the transaction the asset was first seen in, and search joins it against
21+
* `txs` to require a canonical one, so an asset that only ever appeared on an orphaned fork stops
22+
* being returned while the row stays put in case that fork's transaction is later mined.
23+
*
24+
* The backfill resolves a transaction for every asset, so `tx_id` ends up `NOT NULL` and search
25+
* never has to special-case a missing one. No index leads with `asset_identifier` on `ft_events`,
26+
* but none is needed: a holder comes from `ft_balances`, and
27+
* `ft_events_recipient_asset_position_index` (already partial on canonical rows) then resolves
28+
* `(recipient, asset_identifier)` directly. NFT assets go through `nft_events`' `asset_identifier`
29+
* index the same way. Both probe per asset rather than scanning, so the cost is bounded by the
30+
* number of distinct assets rather than by the size of the event tables.
31+
*
32+
* The backfill walks the existing indexes rather than scanning the source tables: a recursive
33+
* "loose index scan" (skip scan) descends once per distinct value, using `ft_balances`'s `(token,
34+
* balance DESC)` index and `nft_custody`'s `(asset_identifier, value)` unique constraint.
35+
* `ft_balances` also stores STX balances under the pseudo-token `stx`, which is not an asset
36+
* identifier, so the backfill keeps only values with the `::` asset separator.
37+
*/
38+
export function up(pgm: MigrationBuilder): void {
39+
pgm.createTable('token_assets', {
40+
asset_identifier: {
41+
type: 'text',
42+
notNull: true,
43+
primaryKey: true,
44+
},
45+
asset_type: {
46+
type: 'text',
47+
notNull: true,
48+
},
49+
tx_id: {
50+
type: 'bytea',
51+
},
52+
});
53+
54+
pgm.sql(`
55+
WITH RECURSIVE tokens AS (
56+
(SELECT token FROM ft_balances ORDER BY token LIMIT 1)
57+
UNION ALL
58+
SELECT (
59+
SELECT b.token FROM ft_balances b WHERE b.token > t.token ORDER BY b.token LIMIT 1
60+
)
61+
FROM tokens t
62+
WHERE t.token IS NOT NULL
63+
)
64+
INSERT INTO token_assets (asset_identifier, asset_type)
65+
SELECT token, 'ft' FROM tokens
66+
WHERE token IS NOT NULL AND token LIKE '%::%'
67+
ON CONFLICT (asset_identifier) DO NOTHING
68+
`);
69+
70+
pgm.sql(`
71+
WITH RECURSIVE assets AS (
72+
(SELECT asset_identifier FROM nft_custody ORDER BY asset_identifier LIMIT 1)
73+
UNION ALL
74+
SELECT (
75+
SELECT c.asset_identifier
76+
FROM nft_custody c
77+
WHERE c.asset_identifier > a.asset_identifier
78+
ORDER BY c.asset_identifier
79+
LIMIT 1
80+
)
81+
FROM assets a
82+
WHERE a.asset_identifier IS NOT NULL
83+
)
84+
INSERT INTO token_assets (asset_identifier, asset_type)
85+
SELECT asset_identifier, 'nft' FROM assets
86+
WHERE asset_identifier IS NOT NULL AND asset_identifier LIKE '%::%'
87+
ON CONFLICT (asset_identifier) DO NOTHING
88+
`);
89+
90+
// Resolve the transaction each asset was first seen in. For fungible tokens that means finding
91+
// any holder, then that holder's receiving event: the holder lookup rides `ft_balances`' `(token,
92+
// balance DESC)` index and the event lookup rides the partial
93+
// `ft_events_recipient_asset_position_index`, so neither scans. Holders whose receiving events
94+
// are all non-canonical drop out of the lateral join, and the next holder is tried.
95+
pgm.sql(`
96+
UPDATE token_assets ta
97+
SET tx_id = (
98+
SELECT e.tx_id
99+
FROM ft_balances b
100+
CROSS JOIN LATERAL (
101+
SELECT ev.tx_id
102+
FROM ft_events ev
103+
WHERE ev.recipient = b.address
104+
AND ev.asset_identifier = ta.asset_identifier
105+
AND ev.canonical = true
106+
AND ev.microblock_canonical = true
107+
LIMIT 1
108+
) e
109+
WHERE b.token = ta.asset_identifier
110+
LIMIT 1
111+
)
112+
WHERE ta.asset_type = 'ft'
113+
`);
114+
115+
pgm.sql(`
116+
UPDATE token_assets ta
117+
SET tx_id = (
118+
SELECT ev.tx_id
119+
FROM nft_events ev
120+
WHERE ev.asset_identifier = ta.asset_identifier
121+
AND ev.canonical = true
122+
AND ev.microblock_canonical = true
123+
LIMIT 1
124+
)
125+
WHERE ta.asset_type = 'nft'
126+
`);
127+
128+
// An asset with no canonical event left is one search would filter out anyway, so drop it rather
129+
// than keep a row that can never be resolved. Its next event re-creates it.
130+
pgm.sql(`DELETE FROM token_assets WHERE tx_id IS NULL`);
131+
pgm.alterColumn('token_assets', 'tx_id', { notNull: true });
132+
133+
// Prefix matching for a partial asset identifier. The primary key's btree uses the database
134+
// collation and cannot serve `LIKE 'term%'`.
135+
pgm.createIndex('token_assets', [{ name: 'asset_identifier', opclass: 'text_pattern_ops' }], {
136+
name: 'token_assets_asset_identifier_pattern_idx',
137+
});
138+
// Only when `pg_trgm` is installed; see `1779800000023_search-name-indexes` for what the API
139+
// falls back to without it. Deferred through `EXECUTE` so `gin_trgm_ops` is never parsed when
140+
// the extension is absent.
141+
pgm.sql(`
142+
DO $$
143+
BEGIN
144+
IF EXISTS (SELECT 1 FROM pg_extension WHERE extname = 'pg_trgm') THEN
145+
EXECUTE 'CREATE INDEX token_assets_asset_identifier_trgm_idx
146+
ON token_assets USING gin (asset_identifier gin_trgm_ops)';
147+
END IF;
148+
END
149+
$$;
150+
`);
151+
}
152+
153+
export function down(pgm: MigrationBuilder): void {
154+
pgm.dropTable('token_assets');
155+
}
Lines changed: 36 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,36 @@
1+
import type { ColumnDefinitions, MigrationBuilder } from 'node-pg-migrate';
2+
3+
export const shorthands: ColumnDefinitions | undefined = undefined;
4+
5+
/**
6+
* Makes the block hash columns answer range scans, so search can resolve a partial block hash.
7+
*
8+
* Both columns carried a hash-method index, which only answers equality — all the API needed while
9+
* it looked blocks up by a complete hash. Matching a pasted hash prefix is a range scan, which only
10+
* a btree can serve. The two columns need different treatment:
11+
*
12+
* - `block_hash` has no other index, so its hash index is replaced with a btree. Equality lookups
13+
* are unaffected in practice: a btree over this table is three levels deep with its upper levels
14+
* permanently cached, so a probe costs the same one or two leaf reads as the hash bucket it
15+
* replaces, and `blocks` takes one insert per Stacks block, far too slow a write rate for the
16+
* extra index maintenance to matter. It costs roughly 0.5–1 GB more on disk, since a btree stores
17+
* the whole 32-byte key where the hash index stored a 4-byte hash code.
18+
* - `index_block_hash` is the table's primary key, so a unique btree already covers it and serves
19+
* range scans. Its hash index is pure duplication once equality no longer needs it, so it is
20+
* dropped rather than replaced, saving both the build and the ongoing maintenance of a second
21+
* full index.
22+
*
23+
* `txs.tx_id` needs no equivalent change: its unique constraint over `(tx_id, index_block_hash,
24+
* microblock_hash)` is already a btree led by `tx_id`.
25+
*/
26+
export function up(pgm: MigrationBuilder): void {
27+
pgm.dropIndex('blocks', 'block_hash');
28+
pgm.createIndex('blocks', 'block_hash');
29+
pgm.dropIndex('blocks', 'index_block_hash');
30+
}
31+
32+
export function down(pgm: MigrationBuilder): void {
33+
pgm.createIndex('blocks', 'index_block_hash', { method: 'hash' });
34+
pgm.dropIndex('blocks', 'block_hash');
35+
pgm.createIndex('blocks', 'block_hash', { method: 'hash' });
36+
}

‎package.json‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -19,6 +19,7 @@
1919
"test:api:principal": "NODE_ENV=test node --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/principal/**/*.test.ts",
2020
"test:api:principal-v3": "NODE_ENV=test node --max-old-space-size=8192 --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/principal-v3/**/*.test.ts",
2121
"test:api:redis": "NODE_ENV=test node --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/redis/**/*.test.ts",
22+
"test:api:search": "NODE_ENV=test node --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/search/**/*.test.ts",
2223
"test:api:smart-contracts": "NODE_ENV=test node --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/smart-contracts/**/*.test.ts",
2324
"test:api:sockets": "NODE_ENV=test node --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/sockets/**/*.test.ts",
2425
"test:api:synthetic-txs": "NODE_ENV=test node --import tsx --test --test-global-setup=./tests/api/setup.ts --test-concurrency=1 ./tests/api/synthetic-txs/**/*.test.ts",

‎src/api/init.ts‎

Lines changed: 4 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -8,7 +8,7 @@ import { CoreNodeRpcProxyRouter } from './routes/v1/core-node-rpc-proxy.js';
88
import { BlockRoutes } from './routes/v1/block.js';
99
import { FaucetRoutes } from './routes/v1/faucets.js';
1010
import { AddressRoutes } from './routes/v1/address.js';
11-
import { SearchRoutes } from './routes/v1/search.js';
11+
import { SearchRoutes as SearchRoutesV1 } from './routes/v1/search.js';
1212
import { StxSupplyRoutes as StxSupplyRoutesV1 } from './routes/v1/stx-supply.js';
1313
import { ChainID } from '../helpers.js';
1414
import { BurnchainRoutes } from './routes/v1/burnchain.js';
@@ -51,6 +51,7 @@ import { BlockTenureRoutes } from './routes/v2/block-tenures.js';
5151
import { PrincipalsRoutes } from './routes/v3/principals.js';
5252
import { TransactionsRoutes } from './routes/v3/transactions.js';
5353
import { MempoolRoutes } from './routes/v3/mempool.js';
54+
import { SearchRoutes } from './routes/v3/search.js';
5455
import { BlocksRoutes } from './routes/v3/blocks.js';
5556
import { StakingBondsRoutes } from './routes/v3/staking-bonds.js';
5657
import { StakingCyclesRoutes } from './routes/v3/staking-cycles.js';
@@ -88,7 +89,7 @@ export const StacksApiRoutes: FastifyPluginAsync<
8889
await fastify.register(BlockRoutes, { prefix: '/block' });
8990
await fastify.register(BurnchainRoutes, { prefix: '/burnchain' });
9091
await fastify.register(AddressRoutes, { prefix: '/address' });
91-
await fastify.register(SearchRoutes, { prefix: '/search' });
92+
await fastify.register(SearchRoutesV1, { prefix: '/search' });
9293
await fastify.register(PoxRoutes, { prefix: '/:pox(pox\\d)' });
9394
await fastify.register(PoxEventRoutes, { prefix: '/:(pox\\d_events)' });
9495
await fastify.register(FaucetRoutes, { prefix: '/faucets' });
@@ -114,6 +115,7 @@ export const StacksApiRoutes: FastifyPluginAsync<
114115
await fastify.register(BlocksRoutes);
115116
await fastify.register(MempoolRoutes);
116117
await fastify.register(PrincipalsRoutes);
118+
await fastify.register(SearchRoutes);
117119
await fastify.register(SmartContractsRoutes);
118120
await fastify.register(StakingBondsRoutes);
119121
await fastify.register(StakingCyclesRoutes);

‎src/api/routes/v1/search.ts‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -257,6 +257,11 @@ export const SearchRoutes: FastifyPluginAsync<
257257
{
258258
schema: {
259259
operationId: 'search_by_id',
260+
deprecated: true,
261+
deprecatedMessage:
262+
'Use GET /extended/v3/search instead. It takes a partial term as well as a complete ' +
263+
'one, matches block heights and token assets in addition to blocks, transactions, ' +
264+
'contracts and accounts, and returns up to 20 ranked matches rather than only the first.',
260265
summary: 'Search',
261266
description: `Search blocks, transactions, contracts, or accounts by hash/ID`,
262267
tags: ['Search'],

0 commit comments

Comments
 (0)