|
| 1 | +import type { ColumnDefinitions, MigrationBuilder } from 'node-pg-migrate'; |
| 2 | + |
| 3 | +export const shorthands: ColumnDefinitions | undefined = undefined; |
| 4 | + |
| 5 | +/** |
| 6 | + * Distinct on-chain token assets, so search can match a term against an asset identifier without |
| 7 | + * scanning the event tables. |
| 8 | + * |
| 9 | + * Asset identifiers are otherwise only recorded per occurrence: `ft_events`/`nft_events` hold one |
| 10 | + * row per transfer and `ft_balances` one row per address/token pair, all of them far too large to |
| 11 | + * substring-match or even to `DISTINCT` on demand. This table holds one row per asset identifier |
| 12 | + * ever seen (order of tens of thousands) which makes a trigram index over it trivially small. |
| 13 | + * |
| 14 | + * Only the on-chain identifier is stored (`SP….contract-name::asset-name`). The display name and |
| 15 | + * symbol a token reports through `get-name`/`get-symbol` are runtime contract-call results that |
| 16 | + * this API never reads; they belong to token-metadata-api. |
| 17 | + * |
| 18 | + * Rows are never removed on a re-org, because a re-org does not re-insert the events that would |
| 19 | + * recreate them: an asset first seen on a block that is later promoted would be lost forever. |
| 20 | + * Instead `tx_id` records the transaction the asset was first seen in, and search joins it against |
| 21 | + * `txs` to require a canonical one, so an asset that only ever appeared on an orphaned fork stops |
| 22 | + * being returned while the row stays put in case that fork's transaction is later mined. |
| 23 | + * |
| 24 | + * The backfill resolves a transaction for every asset, so `tx_id` ends up `NOT NULL` and search |
| 25 | + * never has to special-case a missing one. No index leads with `asset_identifier` on `ft_events`, |
| 26 | + * but none is needed: a holder comes from `ft_balances`, and |
| 27 | + * `ft_events_recipient_asset_position_index` (already partial on canonical rows) then resolves |
| 28 | + * `(recipient, asset_identifier)` directly. NFT assets go through `nft_events`' `asset_identifier` |
| 29 | + * index the same way. Both probe per asset rather than scanning, so the cost is bounded by the |
| 30 | + * number of distinct assets rather than by the size of the event tables. |
| 31 | + * |
| 32 | + * The backfill walks the existing indexes rather than scanning the source tables: a recursive |
| 33 | + * "loose index scan" (skip scan) descends once per distinct value, using `ft_balances`'s `(token, |
| 34 | + * balance DESC)` index and `nft_custody`'s `(asset_identifier, value)` unique constraint. |
| 35 | + * `ft_balances` also stores STX balances under the pseudo-token `stx`, which is not an asset |
| 36 | + * identifier, so the backfill keeps only values with the `::` asset separator. |
| 37 | + */ |
| 38 | +export function up(pgm: MigrationBuilder): void { |
| 39 | + pgm.createTable('token_assets', { |
| 40 | + asset_identifier: { |
| 41 | + type: 'text', |
| 42 | + notNull: true, |
| 43 | + primaryKey: true, |
| 44 | + }, |
| 45 | + asset_type: { |
| 46 | + type: 'text', |
| 47 | + notNull: true, |
| 48 | + }, |
| 49 | + tx_id: { |
| 50 | + type: 'bytea', |
| 51 | + }, |
| 52 | + }); |
| 53 | + |
| 54 | + pgm.sql(` |
| 55 | + WITH RECURSIVE tokens AS ( |
| 56 | + (SELECT token FROM ft_balances ORDER BY token LIMIT 1) |
| 57 | + UNION ALL |
| 58 | + SELECT ( |
| 59 | + SELECT b.token FROM ft_balances b WHERE b.token > t.token ORDER BY b.token LIMIT 1 |
| 60 | + ) |
| 61 | + FROM tokens t |
| 62 | + WHERE t.token IS NOT NULL |
| 63 | + ) |
| 64 | + INSERT INTO token_assets (asset_identifier, asset_type) |
| 65 | + SELECT token, 'ft' FROM tokens |
| 66 | + WHERE token IS NOT NULL AND token LIKE '%::%' |
| 67 | + ON CONFLICT (asset_identifier) DO NOTHING |
| 68 | + `); |
| 69 | + |
| 70 | + pgm.sql(` |
| 71 | + WITH RECURSIVE assets AS ( |
| 72 | + (SELECT asset_identifier FROM nft_custody ORDER BY asset_identifier LIMIT 1) |
| 73 | + UNION ALL |
| 74 | + SELECT ( |
| 75 | + SELECT c.asset_identifier |
| 76 | + FROM nft_custody c |
| 77 | + WHERE c.asset_identifier > a.asset_identifier |
| 78 | + ORDER BY c.asset_identifier |
| 79 | + LIMIT 1 |
| 80 | + ) |
| 81 | + FROM assets a |
| 82 | + WHERE a.asset_identifier IS NOT NULL |
| 83 | + ) |
| 84 | + INSERT INTO token_assets (asset_identifier, asset_type) |
| 85 | + SELECT asset_identifier, 'nft' FROM assets |
| 86 | + WHERE asset_identifier IS NOT NULL AND asset_identifier LIKE '%::%' |
| 87 | + ON CONFLICT (asset_identifier) DO NOTHING |
| 88 | + `); |
| 89 | + |
| 90 | + // Resolve the transaction each asset was first seen in. For fungible tokens that means finding |
| 91 | + // any holder, then that holder's receiving event: the holder lookup rides `ft_balances`' `(token, |
| 92 | + // balance DESC)` index and the event lookup rides the partial |
| 93 | + // `ft_events_recipient_asset_position_index`, so neither scans. Holders whose receiving events |
| 94 | + // are all non-canonical drop out of the lateral join, and the next holder is tried. |
| 95 | + pgm.sql(` |
| 96 | + UPDATE token_assets ta |
| 97 | + SET tx_id = ( |
| 98 | + SELECT e.tx_id |
| 99 | + FROM ft_balances b |
| 100 | + CROSS JOIN LATERAL ( |
| 101 | + SELECT ev.tx_id |
| 102 | + FROM ft_events ev |
| 103 | + WHERE ev.recipient = b.address |
| 104 | + AND ev.asset_identifier = ta.asset_identifier |
| 105 | + AND ev.canonical = true |
| 106 | + AND ev.microblock_canonical = true |
| 107 | + LIMIT 1 |
| 108 | + ) e |
| 109 | + WHERE b.token = ta.asset_identifier |
| 110 | + LIMIT 1 |
| 111 | + ) |
| 112 | + WHERE ta.asset_type = 'ft' |
| 113 | + `); |
| 114 | + |
| 115 | + pgm.sql(` |
| 116 | + UPDATE token_assets ta |
| 117 | + SET tx_id = ( |
| 118 | + SELECT ev.tx_id |
| 119 | + FROM nft_events ev |
| 120 | + WHERE ev.asset_identifier = ta.asset_identifier |
| 121 | + AND ev.canonical = true |
| 122 | + AND ev.microblock_canonical = true |
| 123 | + LIMIT 1 |
| 124 | + ) |
| 125 | + WHERE ta.asset_type = 'nft' |
| 126 | + `); |
| 127 | + |
| 128 | + // An asset with no canonical event left is one search would filter out anyway, so drop it rather |
| 129 | + // than keep a row that can never be resolved. Its next event re-creates it. |
| 130 | + pgm.sql(`DELETE FROM token_assets WHERE tx_id IS NULL`); |
| 131 | + pgm.alterColumn('token_assets', 'tx_id', { notNull: true }); |
| 132 | + |
| 133 | + // Prefix matching for a partial asset identifier. The primary key's btree uses the database |
| 134 | + // collation and cannot serve `LIKE 'term%'`. |
| 135 | + pgm.createIndex('token_assets', [{ name: 'asset_identifier', opclass: 'text_pattern_ops' }], { |
| 136 | + name: 'token_assets_asset_identifier_pattern_idx', |
| 137 | + }); |
| 138 | + // Only when `pg_trgm` is installed; see `1779800000023_search-name-indexes` for what the API |
| 139 | + // falls back to without it. Deferred through `EXECUTE` so `gin_trgm_ops` is never parsed when |
| 140 | + // the extension is absent. |
| 141 | + pgm.sql(` |
| 142 | + DO $$ |
| 143 | + BEGIN |
| 144 | + IF EXISTS (SELECT 1 FROM pg_extension WHERE extname = 'pg_trgm') THEN |
| 145 | + EXECUTE 'CREATE INDEX token_assets_asset_identifier_trgm_idx |
| 146 | + ON token_assets USING gin (asset_identifier gin_trgm_ops)'; |
| 147 | + END IF; |
| 148 | + END |
| 149 | + $$; |
| 150 | + `); |
| 151 | +} |
| 152 | + |
| 153 | +export function down(pgm: MigrationBuilder): void { |
| 154 | + pgm.dropTable('token_assets'); |
| 155 | +} |
0 commit comments