From 234f78493b3f59ed180f078be9d3a12a956fcb5b Mon Sep 17 00:00:00 2001 From: Ivan Vasilov Date: Sat, 26 Sep 2026 21:43:51 +0200 Subject: [PATCH 1/2] feat: split oversized IN filters into multiple requests Reads whose URL would exceed the gateway's ~8 KB request-line limit (414 URI Too Long) now split their largest top-level IN list into chunks that each fit, fetched with bounded concurrency and concatenated. Adds a `maxUrlLength` collection option (defaults to the client's `urlLengthLimit`, 8000). Co-Authored-By: Claude Opus 5.5 (1M context) --- README.md | 9 + src/db.ts | 15 +- src/functions.ts | 180 +++++++++++-- src/postgrest-filters/common.ts | 4 +- src/postgrest-filters/index.ts | 4 + .../split-load-subset-options.ts | 248 ++++++++++++++++++ src/postgrest-request.ts | 9 + tests/e2e/reads.test.ts | 58 +++- tests/pagination.utils.ts | 29 +- tests/split-load-fanout.test.ts | 167 ++++++++++++ tests/split-load-subset-options.test.ts | 240 +++++++++++++++++ 11 files changed, 942 insertions(+), 21 deletions(-) create mode 100644 src/postgrest-filters/split-load-subset-options.ts create mode 100644 tests/split-load-fanout.test.ts create mode 100644 tests/split-load-subset-options.test.ts diff --git a/README.md b/README.md index 0392ee3..4c7d034 100644 --- a/README.md +++ b/README.md @@ -134,6 +134,7 @@ const todos = createCollection( | `realtime` | `boolean` | No | When `true`, subscribes to Postgres changes and reconciles inserts, updates, and deletes into the collection. Defaults to `false`. | | `realtimeUseFilter` | `boolean` | No | **Experimental.** Only applies when `realtime` is `true`. When `true`, each active query's `WHERE` clause is pushed to the Realtime subscription as a `postgres_changes` filter, so the channel only receives changes those queries care about. Defaults to `false`, which subscribes to every change on the table and filters client-side — simpler, at the cost of more Realtime traffic. Queries whose `WHERE` cannot be expressed as a Realtime filter (e.g. `or(...)`) transparently fall back to the unfiltered subscription. | | `queryClient` | `QueryClient` | No | TanStack Query client. If omitted, a shared global client is used. | +| `maxUrlLength` | `number` | No | The longest request-line URL a single read may produce, in characters. A `loadSubset` whose rendered URL would exceed this is split into several requests instead (see [Oversized `IN` lists](#oversized-in-lists) below). Defaults to `supabase.from(tableName).urlLengthLimit` (itself 8000 unless configured on the client). | **Returns** a collection options object to pass to `createCollection`. @@ -200,6 +201,14 @@ Known limitations of the paging loop: - **Concurrent writes.** Offset paging is not keyset-stable: a row inserted or deleted by another client while a multi-page read is in flight shifts the remaining rows by one offset, so a row can be missed or duplicated until the next refetch. - **Count cost.** Every collection read issues one `Prefer: count=exact` on its first request so the loop knows when to stop. On very large tables, or under heavy RLS, that is a full `count(*)` per read. +### Oversized `IN` lists + +Every read is a PostgREST `GET`, so every filter — including a large `inArray(...)`, the shape a lazy join's on-demand collection produces for its join keys — lives in the query string. Supabase's API gateway rejects request lines over about 8 KB with `414 URI Too Long`, and postgrest-js's own `urlLengthLimit` (default 8000) is purely diagnostic: it only adds a hint to the error, it never splits or reroutes the request. + +This adapter does the splitting itself. When a `loadSubset`'s rendered URL would exceed the budget (`maxUrlLength`, see above), it slices the request's largest top-level `inArray(...)` filter into several disjoint requests that each fit, runs them with a concurrency cap, and concatenates the results. This is transparent to the rest of the collection: the query key stays the full, unchunked subset, so one query still owns every row it returns, and `useLiveQuery` code needs no changes. + +Only a positive, top-level AND-ed `inArray(...)` is ever split — never a `not(inArray(...))`, and never one nested inside an `or(...)`, since chunking either would change which rows the union of requests matches. When no such filter exists, or the subset still cannot be made to fit, the request is sent as-is and the server's response (success or error) surfaces exactly as it would without this adapter. + **Evaluated Client-Side** These operations fetch the required rows and process them in memory: diff --git a/src/db.ts b/src/db.ts index ae2ec69..4c74db2 100644 --- a/src/db.ts +++ b/src/db.ts @@ -21,6 +21,17 @@ interface SupabaseCollectionOptions { * update and delete operations. */ keys: Array & string> + /** + * The longest request-line URL a single read is allowed to produce, in + * characters. A `loadSubset` whose rendered URL would exceed this is instead + * split into several requests over disjoint slices of its largest `IN` list + * (run with bounded concurrency and concatenated) so a big `inArray(...)` + * filter — the common shape for on-demand joins — doesn't trip the API + * gateway's request-line limit (Supabase's is about 8 KB) or fail outright. + * Defaults to `supabase.from(tableName).urlLengthLimit` (itself 8000 unless + * configured on the client) when unset. Raise it if your proxy allows more. + */ + maxUrlLength?: number /** The query client */ queryClient?: QueryClient /** Whether to receive updates when a record has been inserted, updated, or deleted by another user */ @@ -97,6 +108,7 @@ export const supabaseCollectionOptions = ({ supabase, realtime, realtimeUseFilter = false, + maxUrlLength, }: SupabaseCollectionOptions) => { // if the query client is not provided, use the global query client queryClient = queryClient ?? getQueryClient() @@ -133,7 +145,8 @@ export const supabaseCollectionOptions = ({ // published. Gating the fetch on the subscription closes this gap but couples // every first load to Realtime connect latency, so it is intentionally left // out and tracked separately. - queryFn: (ctx) => supabaseQueryFn(supabase, tableName, ctx), + queryFn: (ctx) => + supabaseQueryFn(supabase, tableName, ctx, { maxUrlLength }), onInsert: (ctx) => supabaseOnInsert(supabase, tableName, ctx), onUpdate: (ctx) => supabaseOnUpdate(supabase, tableName, keyColumns, ctx), onDelete: (ctx) => supabaseOnDelete(supabase, tableName, keyColumns, ctx), diff --git a/src/functions.ts b/src/functions.ts index 2ebec94..c74c8c1 100644 --- a/src/functions.ts +++ b/src/functions.ts @@ -11,6 +11,7 @@ import { cursorWhereFromToSearch, keyColumnsToSearch, loadSubsetOptionsToSearch, + splitLoadSubsetOptions, subsetParamsToSearch, } from "./postgrest-filters" import { postgrestRequest } from "./postgrest-request" @@ -58,6 +59,8 @@ type PageOptions = { /** The starting offset (always 0 for cursor reads). */ offset: number signal?: AbortSignal + /** Threaded through to postgrest-js's own request-line length diagnostic. */ + urlLengthLimit: number } /** @@ -72,7 +75,7 @@ async function fetchAllPages( supabase: SupabaseClient, tableName: string, baseSearch: URLSearchParams, - { limit, offset: startOffset, signal }: PageOptions + { limit, offset: startOffset, signal, urlLengthLimit }: PageOptions ): Promise { // A zero limit is an empty window: return it without touching the network. if (limit === 0) { @@ -87,6 +90,7 @@ async function fetchAllPages( search: baseSearch, signal, count: "exact", + urlLengthLimit, }) const rows: any[] = first.data ? [...first.data] : [] @@ -109,6 +113,7 @@ async function fetchAllPages( method: "GET", search: pageSearch, signal, + urlLengthLimit, }) const pageRows: any[] = page.data ?? [] rows.push(...pageRows) @@ -118,19 +123,19 @@ async function fetchAllPages( return rows } -export const supabaseQueryFn = async ( +/** + * Read one already-in-budget subset's full window: the keyset tie request (if + * any) plus the main paged request. This is the entire body `supabaseQueryFn` + * used to run directly; it is now also what runs once per chunk when + * `splitLoadSubsetOptions` had to split an oversized subset into several. + */ +async function loadWindow( supabase: SupabaseClient, tableName: string, - ctx: { - client: QueryClient - queryKey: readonly unknown[] - signal: AbortSignal - meta: QueryMeta | undefined - pageParam?: unknown - direction?: unknown - } -) => { - const options = ctx.meta?.loadSubsetOptions ?? {} + options: LoadSubsetOptions, + signal: AbortSignal, + urlLengthLimit: number +) { const search = loadSubsetOptionsToSearch(options) // Cursor reads pin their own window start, so they never carry an offset (see // loadSubsetOptionsToSearch); only a cursor-less read advances from `offset`. @@ -152,7 +157,8 @@ export const supabaseQueryFn = async ( return await fetchAllPages(supabase, tableName, search, { limit: options.limit, offset: startOffset, - signal: ctx.signal, + signal, + urlLengthLimit, }) } @@ -162,12 +168,14 @@ export const supabaseQueryFn = async ( // larger than the cap is not truncated. fetchAllPages(supabase, tableName, tiesSearch, { offset: 0, - signal: ctx.signal, + signal, + urlLengthLimit, }), fetchAllPages(supabase, tableName, search, { limit: options.limit, offset: startOffset, - signal: ctx.signal, + signal, + urlLengthLimit, }), ]) // `whereCurrent` (== boundary) and `whereFrom` (> / < boundary) are disjoint, @@ -175,6 +183,148 @@ export const supabaseQueryFn = async ( return [...ties, ...rows] } +// PostgREST has no body-carried filters for reads and the Supabase API gateway +// rejects request lines over about 8 KB, so a big `inArray(...)` predicate +// (the common shape a lazy join's on-demand collection produces) can make a +// subset's URL too long to send. This default mirrors postgrest-js's own +// `urlLengthLimit` default so, absent an override, the two agree on what +// counts as "too long". +const DEFAULT_URL_LENGTH_LIMIT = 8000 + +// At most this many chunk requests run at once; splitting a large IN list can +// produce far more chunks than are worth having in flight simultaneously. +const MAX_CONCURRENT_CHUNKS = 6 + +/** + * The request-line budget chunks are measured against: the collection's own + * `maxUrlLength` option, else whatever `urlLengthLimit` the table's own + * `PostgrestQueryBuilder` already carries (itself 8000 unless the caller + * configured the Supabase client differently), else the hard-coded default. + * Reading it off `supabase.from(tableName)` keeps this budget in agreement + * with the diagnostic threshold `postgrestRequest` passes to postgrest-js for + * the same table. + */ +function resolveUrlLengthLimit( + supabase: SupabaseClient, + tableName: string, + maxUrlLength: number | undefined +): number { + if (maxUrlLength !== undefined) { + return maxUrlLength + } + const builderLimit = supabase.from(tableName).urlLengthLimit + return typeof builderLimit === "number" + ? builderLimit + : DEFAULT_URL_LENGTH_LIMIT +} + +/** + * The length of the longest URL a subset would produce: the base table URL + * plus the main request's query string, and — when the subset has a cursor — + * the unlimited `whereCurrent` tie request's query string too, since + * `loadWindow` sends both for a cursor read. Measuring rendered URLs (rather + * than estimating) accounts for quoting and percent-encoding exactly. + */ +function measureSubset( + supabase: SupabaseClient, + tableName: string, + options: LoadSubsetOptions +): number { + const base = supabase.from(tableName).url.toString().length + const mainLength = base + loadSubsetOptionsToSearch(options).toString().length + const tiesSearch = cursorCurrentToSearch(options) + return tiesSearch + ? Math.max(mainLength, base + tiesSearch.toString().length) + : mainLength +} + +/** + * Runs one `loadWindow` per chunk with at most `MAX_CONCURRENT_CHUNKS` in + * flight at a time, and concatenates their rows. Chunks are disjoint on the + * split column (see `splitLoadSubsetOptions`), so concatenation order does not + * matter and needs no dedupe. + * + * On the first failure, every other chunk's request is aborted through a + * local `AbortController` combined with the caller's own signal + * (`AbortSignal.any`) — this keeps the same all-or-nothing semantics an + * unsplit request has ("one failed page fails the whole load") instead of + * surfacing a partial result, and the first error encountered is what + * `Promise.all` rejects with. + */ +async function loadChunks( + supabase: SupabaseClient, + tableName: string, + chunks: LoadSubsetOptions[], + signal: AbortSignal, + urlLengthLimit: number +): Promise { + const controller = new AbortController() + const combinedSignal = AbortSignal.any([signal, controller.signal]) + const results: any[][] = new Array(chunks.length) + let nextIndex = 0 + + const runWorker = async (): Promise => { + while (nextIndex < chunks.length) { + const index = nextIndex++ + try { + results[index] = await loadWindow( + supabase, + tableName, + chunks[index], + combinedSignal, + urlLengthLimit + ) + } catch (error) { + controller.abort() + throw error + } + } + } + + const workerCount = Math.min(MAX_CONCURRENT_CHUNKS, chunks.length) + await Promise.all(Array.from({ length: workerCount }, runWorker)) + return results.flat() +} + +export const supabaseQueryFn = async ( + supabase: SupabaseClient, + tableName: string, + ctx: { + client: QueryClient + queryKey: readonly unknown[] + signal: AbortSignal + meta: QueryMeta | undefined + pageParam?: unknown + direction?: unknown + }, + collectionOptions: { maxUrlLength?: number } = {} +) => { + const options = ctx.meta?.loadSubsetOptions ?? {} + const maxUrlLength = resolveUrlLengthLimit( + supabase, + tableName, + collectionOptions.maxUrlLength + ) + // The query key stays the full, unchunked subset (subsetOptionsToQueryKey + // never sees these chunks), so this fan-out is invisible to + // query-db-collection: one query still owns every row it returns. + const chunks = splitLoadSubsetOptions(options, { + maxUrlLength, + measure: (chunkOptions) => measureSubset(supabase, tableName, chunkOptions), + }) + + if (chunks.length === 1) { + return await loadWindow( + supabase, + tableName, + chunks[0], + ctx.signal, + maxUrlLength + ) + } + return await loadChunks(supabase, tableName, chunks, ctx.signal, maxUrlLength) +} + export const supabaseOnInsert = async ( supabase: SupabaseClient, tableName: string, diff --git a/src/postgrest-filters/common.ts b/src/postgrest-filters/common.ts index a12938a..dcf725a 100644 --- a/src/postgrest-filters/common.ts +++ b/src/postgrest-filters/common.ts @@ -52,7 +52,7 @@ export function paramsToSearch(params: PostgrestParam[]): URLSearchParams { return search } -function flattenAnd(expr: Expression): Expression[] { +export function flattenAnd(expr: Expression): Expression[] { return expr.type === "func" && expr.name === "and" ? expr.args.flatMap(flattenAnd) : [expr] @@ -168,7 +168,7 @@ function toFilterString( // Preserve the live adapter's union of duplicate IN demands. This deliberately // fetches a superset, which TanStack re-filters. Never do this inside OR/NOT or // for server aggregates, where client-side filtering cannot repair the result. -function mergeInFilters(filters: Expression[]): Expression[] { +export function mergeInFilters(filters: Expression[]): Expression[] { const merged: Expression[] = [] const valuesByField = new Map() for (const filter of filters) { diff --git a/src/postgrest-filters/index.ts b/src/postgrest-filters/index.ts index efd6683..0ea9fb7 100644 --- a/src/postgrest-filters/index.ts +++ b/src/postgrest-filters/index.ts @@ -6,4 +6,8 @@ export { } from "./load-subset-options-to-search" export { queryIrToSearch } from "./query-ir-to-search" export { realtimeFiltersToSearch } from "./realtime-filters-to-search" +export { + type SplitLoadSubsetOptionsConfig, + splitLoadSubsetOptions, +} from "./split-load-subset-options" export { subsetParamsToSearch } from "./subset-params-to-search" diff --git a/src/postgrest-filters/split-load-subset-options.ts b/src/postgrest-filters/split-load-subset-options.ts new file mode 100644 index 0000000..cd42a07 --- /dev/null +++ b/src/postgrest-filters/split-load-subset-options.ts @@ -0,0 +1,248 @@ +import { IR, type LoadSubsetOptions } from "@tanstack/db" +import type { SerializedExpression as Expression } from "../serialize" +import { flattenAnd, mergeInFilters } from "./common" + +export interface SplitLoadSubsetOptionsConfig { + /** The longest URL (in characters) a single chunk's request may produce. */ + maxUrlLength: number + /** + * Renders the longest URL a given subset would produce (main request, and + * the `whereCurrent` tie request when the subset has a cursor). Injected so + * the splitter stays a pure function of its inputs — the caller supplies the + * real `supabase.from(table).url` + search rendering, tests supply a fake. + */ + measure: (options: LoadSubsetOptions) => number +} + +type SplittableConjunct = { + index: number + path: string[] + values: unknown[] +} + +// Mirrors `renderComparison`'s IN handling (common.ts): TanStack's membership +// test ignores null members, and duplicates contribute nothing extra to the +// rendered list, so both are stripped before we decide how big a conjunct is +// and before we hand out chunk values. +function dedupedValues(values: unknown[]): unknown[] { + return Array.from(new Set(values.filter((value) => value != null))) +} + +// Only a positive, top-level `in(ref, [...])` conjunct is safe to split: the +// union of its chunks is the same set of matching rows. `not(in(...))` and any +// IN nested in OR/NOT are left untouched by the caller (they never reach here +// as a candidate) because a chunked union is not equivalent to their meaning. +function splittableConjuncts(conjuncts: Expression[]): SplittableConjunct[] { + const candidates: SplittableConjunct[] = [] + conjuncts.forEach((conjunct, index) => { + if (conjunct.type !== "func" || conjunct.name !== "in") { + return + } + const [left, right] = conjunct.args + if ( + left?.type !== "ref" || + right?.type !== "val" || + !Array.isArray(right.value) + ) { + return + } + const values = dedupedValues(right.value) + if (values.length === 0) { + // Nothing to chunk (every member was null, or the list was empty); leave + // the conjunct exactly as it was rendered before. + return + } + candidates.push({ index, path: left.path, values }) + }) + // Largest first: the biggest IN list is the one worth splitting, and the + // fallback in `splitConjuncts` halves the next-largest one. + return candidates.sort((a, b) => b.values.length - a.values.length) +} + +function replaceConjunct( + conjuncts: Expression[], + candidate: SplittableConjunct, + values: unknown[] +): Expression[] { + const copy = conjuncts.slice() + copy[candidate.index] = new IR.Func("in", [ + new IR.PropRef(candidate.path), + new IR.Value(values), + ]) + return copy +} + +function rebuildWhere(conjuncts: Expression[]): LoadSubsetOptions["where"] { + if (conjuncts.length === 0) { + return + } + const where = + conjuncts.length === 1 + ? conjuncts[0] + : new IR.Func("and", conjuncts as unknown as IR.BasicExpression[]) + // `conjuncts` entries are either untouched pieces of the caller's own + // `where` (already real IR instances) or `IR.Func`/`IR.Value`/`IR.PropRef` + // instances built above, so this is a widen-then-narrow of the same runtime + // shape `toPostgrestParams` already accepts elsewhere in this package. + return where as unknown as LoadSubsetOptions["where"] +} + +// Applies the caller's `options` (limit/offset/cursor, orderBy, etc.) to one +// chunk's conjunct list, rewriting limit/offset per the plan: a cursor read +// keeps its limit unchanged (the cursor already pins the window start via a +// filter); a cursor-less read with an offset must ask each chunk for enough +// rows to cover the *whole* global window (`limit + offset`) starting at 0, +// because every row in that window ranks at most `offset + limit` inside its +// own chunk — the union of per-chunk windows then still contains the global +// window, with no cross-chunk merge needed since chunks are disjoint on the +// split column. +function buildChunkOptions( + options: LoadSubsetOptions, + conjuncts: Expression[] +): LoadSubsetOptions { + const chunk: LoadSubsetOptions = { + ...options, + where: rebuildWhere(conjuncts), + } + if (!options.cursor && options.offset) { + if (options.limit !== undefined) { + chunk.limit = options.limit + options.offset + } + chunk.offset = undefined + } + return chunk +} + +// Binary-search the longest prefix of `values` whose chunk still fits the +// budget; `fits` is monotonically decreasing in prefix length, so this is +// O(log n) renders instead of O(n). +function longestFittingPrefix( + fits: (length: number) => boolean, + count: number +): number { + if (count === 0 || !fits(1)) { + return 0 + } + if (fits(count)) { + return count + } + let low = 1 + let high = count + while (low < high) { + const mid = low + Math.ceil((high - low) / 2) + if (fits(mid)) { + low = mid + } else { + high = mid - 1 + } + } + return low +} + +/** + * Split one oversized subset's conjunct list into several whose rendered URL + * fits `maxUrlLength`. The largest splittable IN is packed greedily; when not + * even one of its values fits because another IN is also huge, that other IN + * is halved and each half is split again from scratch. Halving (rather than + * packing the other IN as tightly as possible) leaves room for more than one + * value of the largest IN per chunk, so the chunk count stays close to the + * minimum. The resulting conjunct lists are a cartesian product of the split + * candidates and remain pairwise disjoint on the split column(s), so the union + * of their result sets is exactly the original subset's — no dedupe needed. + */ +function splitConjuncts( + options: LoadSubsetOptions, + conjuncts: Expression[], + config: SplitLoadSubsetOptionsConfig +): Expression[][] { + const [current, ...rest] = splittableConjuncts(conjuncts) + if (!current) { + return [conjuncts] + } + + const chunks: Expression[][] = [] + let remaining = current.values + while (remaining.length > 0) { + const fits = (length: number) => + config.measure( + buildChunkOptions( + options, + replaceConjunct(conjuncts, current, remaining.slice(0, length)) + ) + ) <= config.maxUrlLength + const fitLength = longestFittingPrefix(fits, remaining.length) + + if (fitLength > 0) { + chunks.push( + replaceConjunct(conjuncts, current, remaining.slice(0, fitLength)) + ) + remaining = remaining.slice(fitLength) + continue + } + + const other = rest.find((candidate) => candidate.values.length > 1) + if (!other) { + // Nothing left to shrink and even a single value does not fit: take it + // anyway so the loop always makes progress. The request may still come + // back oversized, but that is no worse than not splitting at all — the + // server answers (or errors) exactly as it would without this module. + chunks.push(replaceConjunct(conjuncts, current, remaining.slice(0, 1))) + remaining = remaining.slice(1) + continue + } + + // Another IN is too big to leave room for even one of `current`'s values: + // halve it, and split each half (with `current`'s remaining values) again. + const withRemaining = replaceConjunct(conjuncts, current, remaining) + const middle = Math.ceil(other.values.length / 2) + for (const half of [ + other.values.slice(0, middle), + other.values.slice(middle), + ]) { + chunks.push( + ...splitConjuncts( + options, + replaceConjunct(withRemaining, other, half), + config + ) + ) + } + remaining = [] + } + return chunks +} + +/** + * Split an oversized `loadSubset` request into several whose rendered URL each + * fit `maxUrlLength`, so a caller can run them as separate requests and + * concatenate the results. Returns `[options]` unchanged — the same object, + * not a copy — whenever it already fits, so the common path sends a + * byte-identical URL and keeps today's behaviour. + * + * Only a positive, top-level `AND`ed `in(ref, [...])` conjunct is ever split: + * splitting inside `OR`/`NOT`, or a top-level `not(in(...))`, would change + * which rows the union of chunks matches. When no such conjunct exists (or + * `options.where` is absent), this returns `[options]` and the caller sends + * the request as-is, letting the server's own error (if any) surface exactly + * as it does today. + */ +export function splitLoadSubsetOptions( + options: LoadSubsetOptions, + config: SplitLoadSubsetOptionsConfig +): LoadSubsetOptions[] { + if (config.measure(options) <= config.maxUrlLength) { + return [options] + } + if (!options.where) { + return [options] + } + + const conjuncts = mergeInFilters(flattenAnd(options.where)) + if (splittableConjuncts(conjuncts).length === 0) { + return [options] + } + + return splitConjuncts(options, conjuncts, config).map((chunkConjuncts) => + buildChunkOptions(options, chunkConjuncts) + ) +} diff --git a/src/postgrest-request.ts b/src/postgrest-request.ts index 11c1d00..828c87d 100644 --- a/src/postgrest-request.ts +++ b/src/postgrest-request.ts @@ -23,6 +23,13 @@ interface PostgrestRequestOptions { signal?: AbortSignal /** Expect exactly one row (`Accept: application/vnd.pgrst.object+json`). */ single?: boolean + /** + * The threshold postgrest-js compares the rendered URL's length against + * (default 8000). Purely diagnostic on its side — it only adds a hint to a + * fetch error — but is set to the same budget the adapter's own chunking + * uses, so the two never disagree about what counts as "too long". + */ + urlLengthLimit?: number } /** A PostgREST read result: the rows plus the total count when one was asked for. */ @@ -56,6 +63,7 @@ export async function postgrestRequest( single, count, signal, + urlLengthLimit, }: PostgrestRequestOptions ): Promise { const queryBuilder = supabase.from(table) @@ -85,6 +93,7 @@ export async function postgrestRequest( // The builder threads its own `signal` straight into the underlying fetch, // so aborting the query cancels the in-flight request. signal, + urlLengthLimit, }) const { data, error, count: total } = await builder diff --git a/tests/e2e/reads.test.ts b/tests/e2e/reads.test.ts index 635f199..1d1f737 100644 --- a/tests/e2e/reads.test.ts +++ b/tests/e2e/reads.test.ts @@ -1,4 +1,4 @@ -import { createLiveQueryCollection, eq } from "@tanstack/db" +import { createLiveQueryCollection, eq, inArray } from "@tanstack/db" import { expect, vi } from "vitest" import { queryOnce } from "../../src/index" import { test, WAIT } from "./e2e.utils" @@ -76,3 +76,59 @@ test("reads a set larger than the server row cap across pages", async ({ await live.cleanup() } }) + +// Empirically confirmed against the local stack (Kong): a raw request with +// 1,500 sequential ids renders to ~9.4 KB and comes back `414 Request-URI Too +// Large` with an `error.message` mentioning the URI length. 500-1,200 ids +// (~3-7.3 KB) still succeed, so the failure is specifically the request-line +// limit this feature works around, not some unrelated cap. +test("a raw oversized IN filter is rejected by the gateway (baseline)", async ({ + other, +}) => { + const ids = Array.from({ length: 1500 }, (_, i) => i + 1) + const { data, error, status } = await other + .from("users") + .select() + .in("id", ids) + + expect(data).toBeNull() + expect(status).toBe(414) + expect(error?.message).toMatch(/too long/i) +}) + +test("the same oversized IN filter comes back complete through a collection", async ({ + users, + other, +}) => { + // Seed enough matching rows that the split spans more than one chunk: Alice + // (id 1) and Bob (id 2) from reset_e2e, plus 50 more (ids 3-52). + const extras = Array.from({ length: 50 }, (_, i) => ({ + name: `Extra ${i}`, + email: `extra${i}@test.com`, + active: true, + })) + const { error: insertError } = await other + .from("users") + .insert(extras as unknown as never) + expect(insertError).toBeNull() + + // Same size (and shape) as the raw request above, which the gateway + // rejected outright — this is the request the adapter must split. + const ids = Array.from({ length: 1500 }, (_, i) => i + 1) + const live = createLiveQueryCollection((q) => + q + .from({ row: users.collection }) + .where(({ row }) => inArray(row.id, ids)) + .select(({ row }) => ({ id: row.id, name: row.name })) + ) + + try { + await live.preload() + await vi.waitFor(() => expect(live.size).toBe(52), WAIT) + // Every matching row exactly once, despite coming back over several + // concatenated chunk requests. + expect(new Set(live.toArray.map((row) => row.id)).size).toBe(52) + } finally { + await live.cleanup() + } +}) diff --git a/tests/pagination.utils.ts b/tests/pagination.utils.ts index 40c84ca..bd385b9 100644 --- a/tests/pagination.utils.ts +++ b/tests/pagination.utils.ts @@ -10,11 +10,18 @@ const COMPARATORS: Record boolean> = { eq: (a, b) => a === b, } +// Parses `in.(1,2,3)`'s value half (`(1,2,3)`) into its member list. Numeric +// ids are all the split-chunk tests need, so members are compared as numbers. +function parseInList(value: string): number[] { + const inner = value.replace(/^\(/, "").replace(/\)$/, "") + return inner === "" ? [] : inner.split(",").map(Number) +} + /** * A mock `fetch` that serves a fixture like PostgREST would for the read paths * the pagination adapter emits: single-column `order`, `limit`, `offset`, and - * `col=op.value` scalar filters (`gt`/`gte`/`lt`/`lte`/`eq`). Non-GET requests - * echo their body back so inserts/updates parse. + * `col=op.value` scalar filters (`gt`/`gte`/`lt`/`lte`/`eq`/`in`). Non-GET + * requests echo their body back so inserts/updates parse. * * `cap` simulates PostgREST's `db-max-rows`: no response ever returns more than * `cap` rows, even when a larger `limit` is requested, while the Content-Range @@ -39,6 +46,11 @@ export function makePaginatingFetch( for (const column of columns) { for (const raw of params.getAll(column)) { const [op, value] = raw.split(/\.(.*)/s) + if (op === "in") { + const members = new Set(parseInList(value)) + rows = rows.filter((r) => members.has(Number(r[column]))) + continue + } const compare = COMPARATORS[op] if (compare) { const target = Number(value) @@ -96,3 +108,16 @@ export function getSearches( decodeURIComponent(new URL(String(call[0])).search).replace(/^\?/, "") ) } + +/** + * The rendered (percent-encoded) URL of every GET the mock captured, in + * order — what actually crosses the wire, and so what a request-line budget + * is measured against. + */ +export function getRawUrls( + mockFetch: ReturnType +): string[] { + return mockFetch.mock.calls + .filter((call) => (call[1]?.method ?? "GET") === "GET") + .map((call) => String(call[0])) +} diff --git a/tests/split-load-fanout.test.ts b/tests/split-load-fanout.test.ts new file mode 100644 index 0000000..c9eae82 --- /dev/null +++ b/tests/split-load-fanout.test.ts @@ -0,0 +1,167 @@ +import { createClient } from "@supabase/supabase-js" +import { IR, inArray } from "@tanstack/db" +import { QueryClient } from "@tanstack/query-core" +import { describe, expect, test, vi } from "vitest" +import { supabaseQueryFn } from "../src/functions" +import { + getRawUrls, + getSearches, + makePaginatingFetch, +} from "./pagination.utils" +import { SUPABASE_KEY, SUPABASE_URL } from "./test.utils" + +// Same table shape `makePaginatingFetch` was built for: a bare numeric `id`. +const idRef = new IR.PropRef(["id"]) + +const run = ( + mockFetch: ReturnType | typeof fetch, + ids: number[], + maxUrlLength: number, + signal: AbortSignal = new AbortController().signal +) => { + const supabase = createClient(SUPABASE_URL, SUPABASE_KEY, { + global: { fetch: mockFetch }, + }) + return supabaseQueryFn( + supabase, + "items", + { + client: new QueryClient(), + queryKey: ["items"], + signal, + meta: { loadSubsetOptions: { where: inArray(idRef, ids) } } as never, + }, + { maxUrlLength } + ) +} + +describe("supabaseQueryFn: chunked fan-out for an oversized IN list", () => { + test("many ids produce multiple requests, every URL within budget, and every matching row comes back", async () => { + const ids = Array.from({ length: 40 }, (_, i) => i + 1) + const fixture = ids.map((id) => ({ id })) + const mockFetch = makePaginatingFetch(fixture) + const maxUrlLength = 100 + + const rows = await run(mockFetch, ids, maxUrlLength) + + expect( + (rows as Array<{ id: number }>).map((r) => r.id).sort((a, b) => a - b) + ).toEqual(ids) + const searches = getSearches(mockFetch) + // The list does not fit in one request at this budget, so it was split. + expect(searches.length).toBeGreaterThan(1) + // Every request the mock saw actually carried an `in.(...)` slice of the + // original list, proving the split happened on the `id` column. + for (const search of searches) { + expect(search).toMatch(/id=in\.\(/) + } + // The budget is measured against `base URL + rendered search string` + // (no separator); the real request URL adds one `?` on top of that, so + // the wire length is at most one character longer than the budget. + for (const rawUrl of getRawUrls(mockFetch)) { + expect(rawUrl.length).toBeLessThanOrEqual(maxUrlLength + 1) + } + }) + + test("splitting and the db-max-rows cap compose: a chunk larger than the cap still pages", async () => { + const ids = Array.from({ length: 40 }, (_, i) => i + 1) + const fixture = ids.map((id) => ({ id })) + // Fits ~20 ids per chunk, so two chunks of 20 — each well above the cap. + const maxUrlLength = 145 + const cap = 7 + const mockFetch = makePaginatingFetch(fixture, { cap }) + + const rows = await run(mockFetch, ids, maxUrlLength) + + const returned = (rows as Array<{ id: number }>).map((r) => r.id) + expect(returned.sort((a, b) => a - b)).toEqual(ids) + expect(new Set(returned).size).toBe(ids.length) + // More requests than there are chunks: each chunk's own paging loop ran. + expect(getSearches(mockFetch).length).toBeGreaterThan(2) + }) + + test("one failing chunk rejects the load and aborts its sibling requests", async () => { + const ids = Array.from({ length: 8 }, (_, i) => i + 1) + // Fits two ids per request, so four chunks are issued concurrently + // (well under MAX_CONCURRENT_CHUNKS). + const maxUrlLength = 62 + let abortedCount = 0 + + const failingFetch = vi + .fn() + .mockImplementation((input, init) => { + const params = new URL(String(input)).searchParams + const idFilter = params.get("id") ?? "" + // The chunk holding id 3 fails fast; every other chunk is slow enough + // that the abort raised by the failure reaches it first. + const isFailingChunk = idFilter.includes("3") + const signal = init?.signal as AbortSignal | undefined + + return new Promise((resolve, reject) => { + const settle = () => { + if (isFailingChunk) { + resolve( + new Response(JSON.stringify({ message: "boom" }), { + status: 500, + headers: { "content-type": "application/json" }, + }) + ) + return + } + resolve( + new Response(JSON.stringify([]), { + status: 200, + headers: { + "content-type": "application/json", + "content-range": "*/0", + }, + }) + ) + } + const timer = setTimeout(settle, isFailingChunk ? 0 : 50) + signal?.addEventListener("abort", () => { + clearTimeout(timer) + abortedCount += 1 + reject(new DOMException("Aborted", "AbortError")) + }) + }) + }) + + await expect(run(failingFetch, ids, maxUrlLength)).rejects.toBeDefined() + // Give the aborted siblings' rejection handlers a turn to run. + await new Promise((resolve) => setTimeout(resolve, 60)) + expect(abortedCount).toBeGreaterThan(0) + }) + + test("concurrency never exceeds the fan-out's concurrency cap", async () => { + const ids = Array.from({ length: 30 }, (_, i) => i + 1) + const fixture = ids.map((id) => ({ id })) + // Two ids per request => 15 chunks, comfortably more than the cap. + const maxUrlLength = 62 + const base = makePaginatingFetch(fixture) + + let active = 0 + let peak = 0 + const trackedFetch = vi.fn(async (input, init) => { + active += 1 + peak = Math.max(peak, active) + // Yield so overlapping calls actually overlap instead of resolving + // synchronously one at a time. + await new Promise((resolve) => setTimeout(resolve, 5)) + try { + return await base(input, init) + } finally { + active -= 1 + } + }) + + const rows = await run(trackedFetch, ids, maxUrlLength) + + expect((rows as Array<{ id: number }>).length).toBe(ids.length) + // MAX_CONCURRENT_CHUNKS in src/functions.ts. + expect(peak).toBeLessThanOrEqual(6) + // The cap should actually bind here — otherwise this test would not have + // exercised the concurrency limiter at all. + expect(peak).toBeGreaterThan(1) + }) +}) diff --git a/tests/split-load-subset-options.test.ts b/tests/split-load-subset-options.test.ts new file mode 100644 index 0000000..034b9d4 --- /dev/null +++ b/tests/split-load-subset-options.test.ts @@ -0,0 +1,240 @@ +import type { LoadSubsetOptions } from "@tanstack/db" +import { and, eq, gt, IR, inArray, not, or } from "@tanstack/db" +import { describe, expect, test } from "vitest" +import { + cursorCurrentToSearch, + loadSubsetOptionsToSearch, + splitLoadSubsetOptions, +} from "../src/postgrest-filters" +import { flattenAnd } from "../src/postgrest-filters/common" + +// A short stand-in for `supabase.from(table).url`; only its length matters. +const BASE_URL = "http://localhost:54321/rest/v1/items" + +const id = new IR.PropRef(["id"]) +const other = new IR.PropRef(["other"]) +const active = new IR.PropRef(["active"]) + +// Mirrors `measureSubset` in src/functions.ts, minus the real supabase client: +// renders the same URLs the adapter would send and measures their length. +const measure = (options: LoadSubsetOptions): number => { + const mainLength = + BASE_URL.length + loadSubsetOptionsToSearch(options).toString().length + const tiesSearch = cursorCurrentToSearch(options) + return tiesSearch + ? Math.max(mainLength, BASE_URL.length + tiesSearch.toString().length) + : mainLength +} + +const split = (options: LoadSubsetOptions, maxUrlLength: number) => + splitLoadSubsetOptions(options, { maxUrlLength, measure }) + +// Every chunk's rendered URL (main request, plus the tie request when the +// subset has a cursor) must fit the budget. +const expectEveryChunkFits = ( + chunks: LoadSubsetOptions[], + maxUrlLength: number +) => { + for (const chunk of chunks) { + expect(measure(chunk)).toBeLessThanOrEqual(maxUrlLength) + } +} + +// Collects the values of every top-level `in(ref, [...])` conjunct matching +// `path`, across every chunk — the union a caller reconstructs by +// concatenating each chunk's rows. +const inValuesFor = (chunks: LoadSubsetOptions[], path: string[]): unknown[] => + chunks.flatMap((chunk) => { + if (!chunk.where) return [] + return flattenAnd(chunk.where).flatMap((expr) => { + if (expr.type !== "func" || expr.name !== "in") return [] + const [left, right] = expr.args + if ( + left?.type !== "ref" || + JSON.stringify(left.path) !== JSON.stringify(path) || + right?.type !== "val" || + !Array.isArray(right.value) + ) { + return [] + } + return right.value + }) + }) + +describe("splitLoadSubsetOptions", () => { + test("returns the identical object when the subset already fits", () => { + const options: LoadSubsetOptions = { where: inArray(id, [1, 2, 3]) } + const result = split(options, 8000) + expect(result).toEqual([options]) + expect(result[0]).toBe(options) + }) + + test("splits string ids with commas and quotes, and Dates, so every chunk fits", () => { + const values = [ + 'Alice, "A"', + "Bob (the builder)", + "back\\slash", + new Date("2024-01-01T00:00:00.000Z"), + new Date("2024-06-15T12:30:00.000Z"), + "plain", + "another, one", + ] + const options: LoadSubsetOptions = { where: inArray(id, values) } + const maxUrlLength = BASE_URL.length + 60 + const chunks = split(options, maxUrlLength) + + expect(chunks.length).toBeGreaterThan(1) + expectEveryChunkFits(chunks, maxUrlLength) + expect(new Set(inValuesFor(chunks, ["id"]))).toEqual(new Set(values)) + }) + + test("the union of chunk values equals the deduped, null-free input", () => { + const values = [1, 2, 3, null, 2, 4, 5, null, 6, 7, 8, 9, 10] + const options: LoadSubsetOptions = { where: inArray(id, values) } + const maxUrlLength = BASE_URL.length + 30 + const chunks = split(options, maxUrlLength) + + expect(chunks.length).toBeGreaterThan(1) + expectEveryChunkFits(chunks, maxUrlLength) + const expected = new Set(values.filter((v) => v != null)) + expect(new Set(inValuesFor(chunks, ["id"]))).toEqual(expected) + }) + + test("leaves a top-level not(in) unsplit", () => { + const values = Array.from({ length: 200 }, (_, i) => `id-${i}`) + const options: LoadSubsetOptions = { where: not(inArray(id, values)) } + const result = split(options, BASE_URL.length + 20) + expect(result).toEqual([options]) + expect(result[0]).toBe(options) + }) + + test("leaves an IN nested inside OR unsplit", () => { + const values = Array.from({ length: 200 }, (_, i) => `id-${i}`) + const options: LoadSubsetOptions = { + where: or(inArray(id, values), eq(active, true)), + } + const result = split(options, BASE_URL.length + 20) + expect(result).toEqual([options]) + expect(result[0]).toBe(options) + }) + + test("an unsplittable subset (no where) returns [options] unchanged", () => { + // Nothing to split, yet still "too long" per the fake budget: the + // splitter must not loop or throw, just hand the request back as-is. + const options: LoadSubsetOptions = {} + const result = split(options, -1) + expect(result).toEqual([options]) + expect(result[0]).toBe(options) + }) + + describe("limit/offset/cursor rewriting", () => { + const values = Array.from({ length: 100 }, (_, i) => i) + + test("keeps limit unchanged and does not touch offset when a cursor is present", () => { + const options: LoadSubsetOptions = { + where: inArray(id, values), + limit: 20, + cursor: { whereFrom: gt(id, 5), whereCurrent: eq(id, 5) }, + } + const chunks = split(options, BASE_URL.length + 20) + + expect(chunks.length).toBeGreaterThan(1) + for (const chunk of chunks) { + expect(chunk.limit).toBe(20) + expect(chunk.cursor).toBe(options.cursor) + } + }) + + test("folds offset into limit and drops offset when there is no cursor", () => { + const options: LoadSubsetOptions = { + where: inArray(id, values), + limit: 10, + offset: 5, + } + const chunks = split(options, BASE_URL.length + 20) + + expect(chunks.length).toBeGreaterThan(1) + for (const chunk of chunks) { + expect(chunk.limit).toBe(15) + expect(chunk.offset).toBeUndefined() + } + }) + + test("drops offset without adding a limit when the caller had none", () => { + const options: LoadSubsetOptions = { + where: inArray(id, values), + offset: 5, + } + const chunks = split(options, BASE_URL.length + 20) + + expect(chunks.length).toBeGreaterThan(1) + for (const chunk of chunks) { + expect(chunk.limit).toBeUndefined() + expect(chunk.offset).toBeUndefined() + } + }) + }) + + test("recurses into the next-largest IN when the largest alone cannot shrink enough", () => { + const idValues = Array.from({ length: 60 }, (_, i) => `id-${i}`) + const otherValues = Array.from({ length: 60 }, (_, i) => `other-${i}`) + const options: LoadSubsetOptions = { + where: and(inArray(id, idValues), inArray(other, otherValues)), + } + // Small enough that even a single `id` value alongside the full `other` + // list does not fit, forcing a split of `other` first. + const maxUrlLength = BASE_URL.length + 80 + const chunks = split(options, maxUrlLength) + + expect(chunks.length).toBeGreaterThan(1) + expectEveryChunkFits(chunks, maxUrlLength) + expect(new Set(inValuesFor(chunks, ["id"]))).toEqual(new Set(idValues)) + expect(new Set(inValuesFor(chunks, ["other"]))).toEqual( + new Set(otherValues) + ) + + // Every original (id, other) pair is covered by exactly one chunk's + // cartesian product, and chunks stay disjoint on both split columns. + const pairs = new Set() + for (const chunk of chunks) { + const chunkIds = inValuesFor([chunk], ["id"]) + const chunkOthers = inValuesFor([chunk], ["other"]) + for (const a of chunkIds) { + for (const b of chunkOthers) { + pairs.add(`${a}|${b}`) + } + } + } + expect(pairs.size).toBe(idValues.length * otherValues.length) + // Disjoint: no pair is fetched by more than one chunk. + const chunkPairCount = chunks.reduce( + (sum, chunk) => + sum + + inValuesFor([chunk], ["id"]).length * + inValuesFor([chunk], ["other"]).length, + 0 + ) + expect(chunkPairCount).toBe(pairs.size) + }) + + test("shrinks the next-largest IN against a single value of the largest, not its full list", () => { + const idValues = Array.from({ length: 300 }, (_, i) => `id-${i}`) + const otherValues = Array.from({ length: 300 }, (_, i) => `other-${i}`) + const options: LoadSubsetOptions = { + where: and(inArray(id, idValues), inArray(other, otherValues)), + } + // Neither full list fits alongside the other, but half of `other` does. + // Measuring `other`'s split against the full `id` list would shrink it to + // one value per chunk (~600 requests); measuring against one `id` value + // keeps it to a handful of `other` slices. + const maxUrlLength = BASE_URL.length + 2000 + const chunks = split(options, maxUrlLength) + + expectEveryChunkFits(chunks, maxUrlLength) + expect(new Set(inValuesFor(chunks, ["id"]))).toEqual(new Set(idValues)) + expect(new Set(inValuesFor(chunks, ["other"]))).toEqual( + new Set(otherValues) + ) + expect(chunks.length).toBeLessThan(150) + }) +}) From 866e7876a9b2a857e331283ed870bb271f4438ac Mon Sep 17 00:00:00 2001 From: Ivan Vasilov Date: Sat, 26 Sep 2026 22:06:03 +0200 Subject: [PATCH 2/2] refactor: make the URL length budget a constant Replace the `maxUrlLength` collection option with a fixed `MAX_URL_LENGTH = 8000`, matching postgrest-js's `urlLengthLimit` default. Co-Authored-By: Claude Opus 5.5 (1M context) --- README.md | 3 +- src/db.ts | 15 +--- src/functions.ts | 72 ++++--------------- src/postgrest-request.ts | 9 --- tests/split-load-fanout.test.ts | 121 ++++++++++++++++++++++---------- 5 files changed, 97 insertions(+), 123 deletions(-) diff --git a/README.md b/README.md index 4c7d034..ed7135e 100644 --- a/README.md +++ b/README.md @@ -134,7 +134,6 @@ const todos = createCollection( | `realtime` | `boolean` | No | When `true`, subscribes to Postgres changes and reconciles inserts, updates, and deletes into the collection. Defaults to `false`. | | `realtimeUseFilter` | `boolean` | No | **Experimental.** Only applies when `realtime` is `true`. When `true`, each active query's `WHERE` clause is pushed to the Realtime subscription as a `postgres_changes` filter, so the channel only receives changes those queries care about. Defaults to `false`, which subscribes to every change on the table and filters client-side — simpler, at the cost of more Realtime traffic. Queries whose `WHERE` cannot be expressed as a Realtime filter (e.g. `or(...)`) transparently fall back to the unfiltered subscription. | | `queryClient` | `QueryClient` | No | TanStack Query client. If omitted, a shared global client is used. | -| `maxUrlLength` | `number` | No | The longest request-line URL a single read may produce, in characters. A `loadSubset` whose rendered URL would exceed this is split into several requests instead (see [Oversized `IN` lists](#oversized-in-lists) below). Defaults to `supabase.from(tableName).urlLengthLimit` (itself 8000 unless configured on the client). | **Returns** a collection options object to pass to `createCollection`. @@ -205,7 +204,7 @@ Known limitations of the paging loop: Every read is a PostgREST `GET`, so every filter — including a large `inArray(...)`, the shape a lazy join's on-demand collection produces for its join keys — lives in the query string. Supabase's API gateway rejects request lines over about 8 KB with `414 URI Too Long`, and postgrest-js's own `urlLengthLimit` (default 8000) is purely diagnostic: it only adds a hint to the error, it never splits or reroutes the request. -This adapter does the splitting itself. When a `loadSubset`'s rendered URL would exceed the budget (`maxUrlLength`, see above), it slices the request's largest top-level `inArray(...)` filter into several disjoint requests that each fit, runs them with a concurrency cap, and concatenates the results. This is transparent to the rest of the collection: the query key stays the full, unchunked subset, so one query still owns every row it returns, and `useLiveQuery` code needs no changes. +This adapter does the splitting itself. When a `loadSubset`'s rendered URL would exceed a fixed 8000-character budget (matching postgrest-js's own `urlLengthLimit` default), it slices the request's largest top-level `inArray(...)` filter into several disjoint requests that each fit, runs them with a concurrency cap, and concatenates the results. This is transparent to the rest of the collection: the query key stays the full, unchunked subset, so one query still owns every row it returns, and `useLiveQuery` code needs no changes. Only a positive, top-level AND-ed `inArray(...)` is ever split — never a `not(inArray(...))`, and never one nested inside an `or(...)`, since chunking either would change which rows the union of requests matches. When no such filter exists, or the subset still cannot be made to fit, the request is sent as-is and the server's response (success or error) surfaces exactly as it would without this adapter. diff --git a/src/db.ts b/src/db.ts index 4c74db2..ae2ec69 100644 --- a/src/db.ts +++ b/src/db.ts @@ -21,17 +21,6 @@ interface SupabaseCollectionOptions { * update and delete operations. */ keys: Array & string> - /** - * The longest request-line URL a single read is allowed to produce, in - * characters. A `loadSubset` whose rendered URL would exceed this is instead - * split into several requests over disjoint slices of its largest `IN` list - * (run with bounded concurrency and concatenated) so a big `inArray(...)` - * filter — the common shape for on-demand joins — doesn't trip the API - * gateway's request-line limit (Supabase's is about 8 KB) or fail outright. - * Defaults to `supabase.from(tableName).urlLengthLimit` (itself 8000 unless - * configured on the client) when unset. Raise it if your proxy allows more. - */ - maxUrlLength?: number /** The query client */ queryClient?: QueryClient /** Whether to receive updates when a record has been inserted, updated, or deleted by another user */ @@ -108,7 +97,6 @@ export const supabaseCollectionOptions = ({ supabase, realtime, realtimeUseFilter = false, - maxUrlLength, }: SupabaseCollectionOptions) => { // if the query client is not provided, use the global query client queryClient = queryClient ?? getQueryClient() @@ -145,8 +133,7 @@ export const supabaseCollectionOptions = ({ // published. Gating the fetch on the subscription closes this gap but couples // every first load to Realtime connect latency, so it is intentionally left // out and tracked separately. - queryFn: (ctx) => - supabaseQueryFn(supabase, tableName, ctx, { maxUrlLength }), + queryFn: (ctx) => supabaseQueryFn(supabase, tableName, ctx), onInsert: (ctx) => supabaseOnInsert(supabase, tableName, ctx), onUpdate: (ctx) => supabaseOnUpdate(supabase, tableName, keyColumns, ctx), onDelete: (ctx) => supabaseOnDelete(supabase, tableName, keyColumns, ctx), diff --git a/src/functions.ts b/src/functions.ts index c74c8c1..0ad8671 100644 --- a/src/functions.ts +++ b/src/functions.ts @@ -59,8 +59,6 @@ type PageOptions = { /** The starting offset (always 0 for cursor reads). */ offset: number signal?: AbortSignal - /** Threaded through to postgrest-js's own request-line length diagnostic. */ - urlLengthLimit: number } /** @@ -75,7 +73,7 @@ async function fetchAllPages( supabase: SupabaseClient, tableName: string, baseSearch: URLSearchParams, - { limit, offset: startOffset, signal, urlLengthLimit }: PageOptions + { limit, offset: startOffset, signal }: PageOptions ): Promise { // A zero limit is an empty window: return it without touching the network. if (limit === 0) { @@ -90,7 +88,6 @@ async function fetchAllPages( search: baseSearch, signal, count: "exact", - urlLengthLimit, }) const rows: any[] = first.data ? [...first.data] : [] @@ -113,7 +110,6 @@ async function fetchAllPages( method: "GET", search: pageSearch, signal, - urlLengthLimit, }) const pageRows: any[] = page.data ?? [] rows.push(...pageRows) @@ -133,8 +129,7 @@ async function loadWindow( supabase: SupabaseClient, tableName: string, options: LoadSubsetOptions, - signal: AbortSignal, - urlLengthLimit: number + signal: AbortSignal ) { const search = loadSubsetOptionsToSearch(options) // Cursor reads pin their own window start, so they never carry an offset (see @@ -158,7 +153,6 @@ async function loadWindow( limit: options.limit, offset: startOffset, signal, - urlLengthLimit, }) } @@ -169,13 +163,11 @@ async function loadWindow( fetchAllPages(supabase, tableName, tiesSearch, { offset: 0, signal, - urlLengthLimit, }), fetchAllPages(supabase, tableName, search, { limit: options.limit, offset: startOffset, signal, - urlLengthLimit, }), ]) // `whereCurrent` (== boundary) and `whereFrom` (> / < boundary) are disjoint, @@ -184,40 +176,16 @@ async function loadWindow( } // PostgREST has no body-carried filters for reads and the Supabase API gateway -// rejects request lines over about 8 KB, so a big `inArray(...)` predicate -// (the common shape a lazy join's on-demand collection produces) can make a -// subset's URL too long to send. This default mirrors postgrest-js's own -// `urlLengthLimit` default so, absent an override, the two agree on what -// counts as "too long". -const DEFAULT_URL_LENGTH_LIMIT = 8000 +// rejects request lines over about 8 KB with 414, so a big `inArray(...)` +// predicate (the common shape a lazy join's on-demand collection produces) +// can make a subset's URL too long to send. This matches postgrest-js's own +// `urlLengthLimit` default so the two agree on what counts as "too long". +export const MAX_URL_LENGTH = 8000 // At most this many chunk requests run at once; splitting a large IN list can // produce far more chunks than are worth having in flight simultaneously. const MAX_CONCURRENT_CHUNKS = 6 -/** - * The request-line budget chunks are measured against: the collection's own - * `maxUrlLength` option, else whatever `urlLengthLimit` the table's own - * `PostgrestQueryBuilder` already carries (itself 8000 unless the caller - * configured the Supabase client differently), else the hard-coded default. - * Reading it off `supabase.from(tableName)` keeps this budget in agreement - * with the diagnostic threshold `postgrestRequest` passes to postgrest-js for - * the same table. - */ -function resolveUrlLengthLimit( - supabase: SupabaseClient, - tableName: string, - maxUrlLength: number | undefined -): number { - if (maxUrlLength !== undefined) { - return maxUrlLength - } - const builderLimit = supabase.from(tableName).urlLengthLimit - return typeof builderLimit === "number" - ? builderLimit - : DEFAULT_URL_LENGTH_LIMIT -} - /** * The length of the longest URL a subset would produce: the base table URL * plus the main request's query string, and — when the subset has a cursor — @@ -255,8 +223,7 @@ async function loadChunks( supabase: SupabaseClient, tableName: string, chunks: LoadSubsetOptions[], - signal: AbortSignal, - urlLengthLimit: number + signal: AbortSignal ): Promise { const controller = new AbortController() const combinedSignal = AbortSignal.any([signal, controller.signal]) @@ -271,8 +238,7 @@ async function loadChunks( supabase, tableName, chunks[index], - combinedSignal, - urlLengthLimit + combinedSignal ) } catch (error) { controller.abort() @@ -296,33 +262,21 @@ export const supabaseQueryFn = async ( meta: QueryMeta | undefined pageParam?: unknown direction?: unknown - }, - collectionOptions: { maxUrlLength?: number } = {} + } ) => { const options = ctx.meta?.loadSubsetOptions ?? {} - const maxUrlLength = resolveUrlLengthLimit( - supabase, - tableName, - collectionOptions.maxUrlLength - ) // The query key stays the full, unchunked subset (subsetOptionsToQueryKey // never sees these chunks), so this fan-out is invisible to // query-db-collection: one query still owns every row it returns. const chunks = splitLoadSubsetOptions(options, { - maxUrlLength, + maxUrlLength: MAX_URL_LENGTH, measure: (chunkOptions) => measureSubset(supabase, tableName, chunkOptions), }) if (chunks.length === 1) { - return await loadWindow( - supabase, - tableName, - chunks[0], - ctx.signal, - maxUrlLength - ) + return await loadWindow(supabase, tableName, chunks[0], ctx.signal) } - return await loadChunks(supabase, tableName, chunks, ctx.signal, maxUrlLength) + return await loadChunks(supabase, tableName, chunks, ctx.signal) } export const supabaseOnInsert = async ( diff --git a/src/postgrest-request.ts b/src/postgrest-request.ts index 828c87d..11c1d00 100644 --- a/src/postgrest-request.ts +++ b/src/postgrest-request.ts @@ -23,13 +23,6 @@ interface PostgrestRequestOptions { signal?: AbortSignal /** Expect exactly one row (`Accept: application/vnd.pgrst.object+json`). */ single?: boolean - /** - * The threshold postgrest-js compares the rendered URL's length against - * (default 8000). Purely diagnostic on its side — it only adds a hint to a - * fetch error — but is set to the same budget the adapter's own chunking - * uses, so the two never disagree about what counts as "too long". - */ - urlLengthLimit?: number } /** A PostgREST read result: the rows plus the total count when one was asked for. */ @@ -63,7 +56,6 @@ export async function postgrestRequest( single, count, signal, - urlLengthLimit, }: PostgrestRequestOptions ): Promise { const queryBuilder = supabase.from(table) @@ -93,7 +85,6 @@ export async function postgrestRequest( // The builder threads its own `signal` straight into the underlying fetch, // so aborting the query cancels the in-flight request. signal, - urlLengthLimit, }) const { data, error, count: total } = await builder diff --git a/tests/split-load-fanout.test.ts b/tests/split-load-fanout.test.ts index c9eae82..e1945b6 100644 --- a/tests/split-load-fanout.test.ts +++ b/tests/split-load-fanout.test.ts @@ -1,8 +1,9 @@ import { createClient } from "@supabase/supabase-js" -import { IR, inArray } from "@tanstack/db" +import { IR, inArray, type LoadSubsetOptions } from "@tanstack/db" import { QueryClient } from "@tanstack/query-core" import { describe, expect, test, vi } from "vitest" -import { supabaseQueryFn } from "../src/functions" +import { MAX_URL_LENGTH, supabaseQueryFn } from "../src/functions" +import { loadSubsetOptionsToSearch } from "../src/postgrest-filters" import { getRawUrls, getSearches, @@ -13,36 +14,68 @@ import { SUPABASE_KEY, SUPABASE_URL } from "./test.utils" // Same table shape `makePaginatingFetch` was built for: a bare numeric `id`. const idRef = new IR.PropRef(["id"]) +// The base table URL `supabaseQueryFn` measures every subset's rendered URL +// against (see `measureSubset` in src/functions.ts) — computed once here so +// `idsExceedingLength` can size an id list against the same budget the +// production splitter uses. +const BASE_URL_LENGTH = createClient(SUPABASE_URL, SUPABASE_KEY) + .from("items") + .url.toString().length + +/** + * A dense `1..n` id list whose rendered `in(...)` filter, combined with the + * base table URL, is at least `minLength` characters — i.e. the URL a single, + * unsplit request for this list would produce. Sized against `MAX_URL_LENGTH` + * (rather than a hard-coded id count) so the tests below keep exercising the + * splitter even if that constant ever changes. + * + * A small sample list's rendered length estimates the average per-id cost + * (digits plus a percent-encoded comma); the list is then resized to that + * estimate. One correction pass is always enough — the only thing the sample + * can get wrong is the id width, which grows by at most a digit or two + * between the sample and the final count. + */ +function idsExceedingLength(minLength: number): number[] { + let count = 100 + for (let attempt = 0; attempt < 10; attempt++) { + const ids = Array.from({ length: count }, (_, i) => i + 1) + const length = + BASE_URL_LENGTH + + loadSubsetOptionsToSearch({ + where: inArray(idRef, ids), + } as unknown as LoadSubsetOptions).toString().length + if (length >= minLength) { + return ids + } + const perId = (length - BASE_URL_LENGTH) / count + count = Math.ceil((minLength - BASE_URL_LENGTH) / perId) + 10 + } + throw new Error(`could not size an id list past ${minLength} characters`) +} + const run = ( mockFetch: ReturnType | typeof fetch, ids: number[], - maxUrlLength: number, signal: AbortSignal = new AbortController().signal ) => { const supabase = createClient(SUPABASE_URL, SUPABASE_KEY, { global: { fetch: mockFetch }, }) - return supabaseQueryFn( - supabase, - "items", - { - client: new QueryClient(), - queryKey: ["items"], - signal, - meta: { loadSubsetOptions: { where: inArray(idRef, ids) } } as never, - }, - { maxUrlLength } - ) + return supabaseQueryFn(supabase, "items", { + client: new QueryClient(), + queryKey: ["items"], + signal, + meta: { loadSubsetOptions: { where: inArray(idRef, ids) } } as never, + }) } describe("supabaseQueryFn: chunked fan-out for an oversized IN list", () => { test("many ids produce multiple requests, every URL within budget, and every matching row comes back", async () => { - const ids = Array.from({ length: 40 }, (_, i) => i + 1) + const ids = idsExceedingLength(MAX_URL_LENGTH * 1.5) const fixture = ids.map((id) => ({ id })) const mockFetch = makePaginatingFetch(fixture) - const maxUrlLength = 100 - const rows = await run(mockFetch, ids, maxUrlLength) + const rows = await run(mockFetch, ids) expect( (rows as Array<{ id: number }>).map((r) => r.id).sort((a, b) => a - b) @@ -55,36 +88,37 @@ describe("supabaseQueryFn: chunked fan-out for an oversized IN list", () => { for (const search of searches) { expect(search).toMatch(/id=in\.\(/) } - // The budget is measured against `base URL + rendered search string` - // (no separator); the real request URL adds one `?` on top of that, so - // the wire length is at most one character longer than the budget. + // Every rendered request line stays within the fixed budget. for (const rawUrl of getRawUrls(mockFetch)) { - expect(rawUrl.length).toBeLessThanOrEqual(maxUrlLength + 1) + expect(rawUrl.length).toBeLessThanOrEqual(MAX_URL_LENGTH + 1) } }) test("splitting and the db-max-rows cap compose: a chunk larger than the cap still pages", async () => { - const ids = Array.from({ length: 40 }, (_, i) => i + 1) + const ids = idsExceedingLength(MAX_URL_LENGTH * 1.5) const fixture = ids.map((id) => ({ id })) - // Fits ~20 ids per chunk, so two chunks of 20 — each well above the cap. - const maxUrlLength = 145 - const cap = 7 + // Small enough that each chunk (hundreds to low thousands of ids) needs + // several pages, but not so small that the paging loop needs thousands of + // requests to get through a chunk. + const cap = 100 const mockFetch = makePaginatingFetch(fixture, { cap }) - const rows = await run(mockFetch, ids, maxUrlLength) + const rows = await run(mockFetch, ids) const returned = (rows as Array<{ id: number }>).map((r) => r.id) expect(returned.sort((a, b) => a - b)).toEqual(ids) expect(new Set(returned).size).toBe(ids.length) + const searches = getSearches(mockFetch) // More requests than there are chunks: each chunk's own paging loop ran. - expect(getSearches(mockFetch).length).toBeGreaterThan(2) + expect(searches.length).toBeGreaterThan(2) }) test("one failing chunk rejects the load and aborts its sibling requests", async () => { - const ids = Array.from({ length: 8 }, (_, i) => i + 1) - // Fits two ids per request, so four chunks are issued concurrently - // (well under MAX_CONCURRENT_CHUNKS). - const maxUrlLength = 62 + const ids = idsExceedingLength(MAX_URL_LENGTH * 3) + // Any id lands in exactly one chunk (chunks are disjoint slices of the + // sorted list), so failing whichever chunk holds this one exercises the + // abort path without depending on how many chunks came out. + const failId = ids[0] let abortedCount = 0 const failingFetch = vi @@ -92,9 +126,14 @@ describe("supabaseQueryFn: chunked fan-out for an oversized IN list", () => { .mockImplementation((input, init) => { const params = new URL(String(input)).searchParams const idFilter = params.get("id") ?? "" - // The chunk holding id 3 fails fast; every other chunk is slow enough - // that the abort raised by the failure reaches it first. - const isFailingChunk = idFilter.includes("3") + const members = idFilter + .replace(/^in\.\(/, "") + .replace(/\)$/, "") + .split(",") + .map(Number) + // The chunk holding `failId` fails fast; every other chunk is slow + // enough that the abort raised by the failure reaches it first. + const isFailingChunk = members.includes(failId) const signal = init?.signal as AbortSignal | undefined return new Promise((resolve, reject) => { @@ -127,17 +166,18 @@ describe("supabaseQueryFn: chunked fan-out for an oversized IN list", () => { }) }) - await expect(run(failingFetch, ids, maxUrlLength)).rejects.toBeDefined() + await expect(run(failingFetch, ids)).rejects.toBeDefined() // Give the aborted siblings' rejection handlers a turn to run. await new Promise((resolve) => setTimeout(resolve, 60)) expect(abortedCount).toBeGreaterThan(0) }) test("concurrency never exceeds the fan-out's concurrency cap", async () => { - const ids = Array.from({ length: 30 }, (_, i) => i + 1) + // MAX_CONCURRENT_CHUNKS (src/functions.ts) is 6; size the list well past + // that so the cap actually binds instead of every chunk fitting in one + // wave. + const ids = idsExceedingLength(MAX_URL_LENGTH * 10) const fixture = ids.map((id) => ({ id })) - // Two ids per request => 15 chunks, comfortably more than the cap. - const maxUrlLength = 62 const base = makePaginatingFetch(fixture) let active = 0 @@ -155,9 +195,12 @@ describe("supabaseQueryFn: chunked fan-out for an oversized IN list", () => { } }) - const rows = await run(trackedFetch, ids, maxUrlLength) + const rows = await run(trackedFetch, ids) expect((rows as Array<{ id: number }>).length).toBe(ids.length) + const searches = getSearches(trackedFetch) + // Sizing above actually produced more chunks than the concurrency cap. + expect(searches.length).toBeGreaterThan(6) // MAX_CONCURRENT_CHUNKS in src/functions.ts. expect(peak).toBeLessThanOrEqual(6) // The cap should actually bind here — otherwise this test would not have