Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 21 additions & 1 deletion eng/profiler_benchmarks/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -12,13 +12,33 @@ python -m eng.profiler_benchmarks.controller --base main --candidate HEAD \
python -m eng.profiler_benchmarks.report profiler-results/report.json
```

The fixed registry has 21 tasks. `--scenarios` runs a local subset, but subset
The fixed registry has 24 tasks. `--scenarios` runs a local subset, but subset
reports remain incomplete and cannot produce a verdict.

`lob_varchar_256k_fetchall` fetches one 256 KiB `VARCHAR(MAX)` value to exercise
multi-chunk streaming. Query setup and exact payload validation are outside the
timed fetch window.

`catalog_columns_2`, `catalog_columns_118`, and `catalog_columns_2111` time
`columns()` plus the complete `fetchall()` drain on a fresh cursor. Each task
creates UUID-prefixed tables in the current database's `dbo` schema, with exactly
2, 118, or 2,111 nullable `INT` columns in total (at most 704 per table). It needs
permission to create tables and uses the profiler-owned connection with autocommit
off, not a caller's connection with pending writes. Its DDL is uncommitted and
rolled back on success or failure. Fixture setup, cursor creation, EOF checks, and
comparison of all 29 provider fields, raw descriptions, ordered cells and Python
types with an independent rowwise drain are outside the measurement window.
The rowwise oracle runs after timing,
so it does not prime the measured catalog allocation. Native call counters verify
one `SQLColumns` and one `FetchAll` call in the measured window.

These tasks cover small results, growth through both initial tiers, and reuse of
the maximum tier. They measure instrumented latency, not allocation bytes, and
are not the same fixtures as the standalone catalog experiments. Ordinary SELECT
fetch-all remains a separate control. The unchanged advisory thresholds below can
leave smaller catalog slowdowns labeled "no signal"; that is not proof of no
regression.

## Measurement contract

CI uses the PR merge's first parent as the exact base. It reuses the
Expand Down
3 changes: 3 additions & 0 deletions eng/profiler_benchmarks/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,9 @@
"fetch_1_2m": "1.2-million-row fetching",
"cte": "Common table expression queries",
"lob_varchar_256k_fetchall": "256 KiB VARCHAR(MAX) / fetchall()",
"catalog_columns_2": "Catalog columns / 2 rows",
"catalog_columns_118": "Catalog columns / 118 rows",
"catalog_columns_2111": "Catalog columns / 2,111 rows",
}
CASES = tuple(TASK_NAMES)
MAX_BYTES = 8 * 1024 * 1024
Expand Down
77 changes: 77 additions & 0 deletions eng/profiler_benchmarks/workloads.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

from functools import partial
import time
from uuid import uuid4

from profiler import scenarios

Expand Down Expand Up @@ -163,6 +164,78 @@ def lob_fetch(conn, ctx):
ctx.disable()


def catalog_columns(conn, ctx, row_count):
"""Use the profiler-owned transaction; time only SQLColumns and its complete drain."""
if conn.autocommit:
raise ValueError("Catalog benchmarks require a transactional connection")
prefix = "profcat" + uuid4().hex
tables = [
(f"{prefix}{index:02d}", min(704, row_count - start))
for index, start in enumerate(range(0, row_count, 704))
]
expected = [
(table, f"c{column:04d}", column + 1) for table, count in tables for column in range(count)
]
try:
with conn.cursor() as setup:
catalog = setup.execute("SELECT DB_NAME()").fetchone()[0]
for table, count in tables:
definitions = ", ".join(f"[c{column:04d}] INT NULL" for column in range(count))
setup.execute(f"CREATE TABLE [dbo].[{table}] ({definitions})")
filters = dict(catalog=catalog, schema="dbo", table=prefix + "%", column=None)
with conn.cursor() as cursor:
try:
ctx.enable()
start = time.perf_counter()
cursor.columns(**filters)
rows = cursor.fetchall()
wall_ms = (time.perf_counter() - start) * 1000
cpp, py = ctx.collect()
finally:
ctx.disable()
description = cursor.description
assert len(rows) == row_count
assert len(description) == 29 and all(len(row) == 29 for row in rows)
assert [
(row.table_name, row.column_name, row.ordinal_position) for row in rows
] == expected
assert all(
row.table_cat == catalog
and row.table_schem == "dbo"
and type(row.data_type) is int
and row.data_type == 4 # SQL_INTEGER
and row.column_size == 10
and row.nullable == 1
and row.column_def is None
for row in rows
)
assert cursor.fetchone() is None
assert not cursor.messages, "Clean catalog fetch unexpectedly produced diagnostics"
# A separate rowwise oracle runs only after the measured first allocation.
with conn.cursor() as reference:
reference.columns(**filters)
assert type(reference.description) is type(description)
assert reference.description == description
expected_rows = [tuple(row) for row in reference]
assert [tuple(row) for row in rows] == expected_rows
assert [[type(value) for value in row] for row in rows] == [
[type(value) for value in row] for row in expected_rows
]
assert not reference.messages, "Rowwise catalog oracle produced diagnostics"
assert cpp["ddbc::SQLColumns_wrap"]["calls"] == 1
assert cpp["ddbc::FetchAll_wrap"]["calls"] == 1
Comment on lines +225 to +226
return dict(
title="SQLColumns metadata",
wall_ms=wall_ms,
cpp=cpp,
py=py,
detail=f"Rows: {row_count}; tables: {len(tables)}; type: INT NULL; API: columns+fetchall",
)
finally:
# Fixture DDL is never committed; rollback also removes partially created fixtures.
conn.rollback()


def registry():
"""Keep every PR #552 scenario, including its existing timing boundaries."""
result = dict(scenarios.SCENARIOS)
Expand All @@ -176,4 +249,8 @@ def registry():
)
result.update((name, (partial(query, sql=sql), False)) for name, sql in QUERIES.items())
result["lob_varchar_256k_fetchall"] = (lob_fetch, False)
result.update(
(f"catalog_columns_{count}", (partial(catalog_columns, row_count=count), False))
for count in (2, 118, 2111)
)
return result
80 changes: 49 additions & 31 deletions mssql_python/pybind/ddbc_bindings.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1886,7 +1886,9 @@ SQLRETURN SQLColumns_wrap(SqlHandlePtr StatementHandle, const py::object& catalo
const py::object& schemaObj, const py::object& tableObj,
const py::object& columnObj) {
PERF_TIMER("SQLColumns_wrap");
StatementHandle->resultMetadata.clear();
StatementHandle->resultMetadata.clear(true);
SQLRETURN ret = SQL_ERROR;
ResultMetadataFailureGuard metadataFailure(StatementHandle->resultMetadata, ret);
if (!SQLColumns_ptr) {
ThrowStdException("SQLColumns function not loaded");
}
Expand All @@ -1897,16 +1899,19 @@ SQLRETURN SQLColumns_wrap(SqlHandlePtr StatementHandle, const py::object& catalo
std::u16string column = columnObj.is_none() ? u"" : columnObj.cast<std::u16string>();

// Release the GIL during the blocking ODBC catalog call
py::gil_scoped_release release;
return SQLColumns_ptr(StatementHandle->get(),
catalog.empty() ? nullptr : reinterpretU16stringAsSqlWChar(catalog),
catalog.empty() ? 0 : SQL_NTS,
schema.empty() ? nullptr : reinterpretU16stringAsSqlWChar(schema),
schema.empty() ? 0 : SQL_NTS,
table.empty() ? nullptr : reinterpretU16stringAsSqlWChar(table),
table.empty() ? 0 : SQL_NTS,
column.empty() ? nullptr : reinterpretU16stringAsSqlWChar(column),
column.empty() ? 0 : SQL_NTS);
{
py::gil_scoped_release release;
ret = SQLColumns_ptr(StatementHandle->get(),
catalog.empty() ? nullptr : reinterpretU16stringAsSqlWChar(catalog),
catalog.empty() ? 0 : SQL_NTS,
schema.empty() ? nullptr : reinterpretU16stringAsSqlWChar(schema),
schema.empty() ? 0 : SQL_NTS,
table.empty() ? nullptr : reinterpretU16stringAsSqlWChar(table),
table.empty() ? 0 : SQL_NTS,
column.empty() ? nullptr : reinterpretU16stringAsSqlWChar(column),
column.empty() ? 0 : SQL_NTS);
}
return ret;
}

// Helper function to check for driver errors
Expand Down Expand Up @@ -6232,32 +6237,45 @@ SQLRETURN FetchAll_wrap(SqlHandlePtr StatementHandle, py::list& rows,
} else {
fetchSize = 1000;
}
LOG("FetchAll_wrap: Fetching data in batch sizes of %d", fetchSize);

ColumnBuffers buffers(numCols, fetchSize);
SQLULEN numRowsFetched = 0;
FetchStateGuard fetchStateGuard(StatementHandle, messages);

// Bind columns
ret = SQLBindColums(hStmt, buffers, columnNames, numCols, fetchSize, charCtype, messages);
if (!SQL_SUCCEEDED(ret)) {
LOG("FetchAll_wrap: Error when binding columns - SQLRETURN=%d", ret);
return ret;
const int maxFetchSize = fetchSize;
// SQLColumns declares wide fields even for small catalogs. Start at the
// existing 10-row tier, then grow through the same tiers as batches fill.
if (metadataSnapshot.catalogResult) {
fetchSize = std::min(fetchSize, 10);
Comment on lines +6243 to +6244
}

fetchStateGuard.configure(&numRowsFetched, fetchSize);

while (ret != SQL_NO_DATA) {
ret = FetchBatchData(hStmt, buffers, columnNames, rows, numCols, numRowsFetched, lobColumns,
charEncoding, charCtype, messages);
CheckFetchError(StatementHandle, ret);
if (!SQL_SUCCEEDED(ret) && ret != SQL_NO_DATA) {
LOG("FetchAll_wrap: Error when fetching data - SQLRETURN=%d", ret);
LOG("FetchAll_wrap: Fetching data in batch sizes of %d", fetchSize);
ColumnBuffers buffers(numCols, fetchSize);
SQLULEN numRowsFetched = 0;
FetchStateGuard fetchStateGuard(StatementHandle, messages);

ret = SQLBindColums(hStmt, buffers, columnNames, numCols, fetchSize, charCtype, messages);
if (!SQL_SUCCEEDED(ret)) {
LOG("FetchAll_wrap: Error when binding columns - SQLRETURN=%d", ret);
return ret;
}
}

fetchStateGuard.close();
fetchStateGuard.configure(&numRowsFetched, fetchSize);

while (ret != SQL_NO_DATA) {
ret = FetchBatchData(hStmt, buffers, columnNames, rows, numCols, numRowsFetched,
lobColumns, charEncoding, charCtype, messages);
CheckFetchError(StatementHandle, ret);
if (!SQL_SUCCEEDED(ret) && ret != SQL_NO_DATA) {
LOG("FetchAll_wrap: Error when fetching data - SQLRETURN=%d", ret);
return ret;
}
if (SQL_SUCCEEDED(ret) && numRowsFetched == static_cast<SQLULEN>(fetchSize) &&
fetchSize < maxFetchSize) {
break;
}
}

// Unbind while these buffers are still alive, before allocating the next tier.
fetchStateGuard.close();
Comment on lines +6275 to +6276
fetchSize = std::min(fetchSize * 10, maxFetchSize);
}

return ret;
}
Expand Down
7 changes: 5 additions & 2 deletions mssql_python/pybind/result_metadata.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -30,11 +30,12 @@ class ResultMetadataCache {
struct Snapshot {
uint64_t generation;
std::shared_ptr<const ResultMetadata> metadata;
bool catalogResult; // SQLColumns allocation hint, reset with the result generation.
};

Snapshot snapshot() const {
std::lock_guard<std::mutex> lock(mutex_);
return {generation_, metadata_};
return {generation_, metadata_, catalogResult_};
}

void publish(uint64_t generation, std::shared_ptr<const ResultMetadata> metadata) {
Expand All @@ -44,16 +45,18 @@ class ResultMetadataCache {
}
}

void clear() {
void clear(bool catalogResult = false) {
std::lock_guard<std::mutex> lock(mutex_);
++generation_;
metadata_.reset();
catalogResult_ = catalogResult;
}

private:
mutable std::mutex mutex_;
uint64_t generation_ = 0;
std::shared_ptr<const ResultMetadata> metadata_;
bool catalogResult_ = false;
};

class ResultMetadataFailureGuard {
Expand Down
68 changes: 68 additions & 0 deletions tests/test_004_cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -13729,6 +13729,10 @@ def test_columns_specific_table(cursor, db_connection):

# Verify we got results
assert len(cols) == 9, "Should find exactly 9 columns in columns_test"
assert [col.ordinal_position for col in cols] == list(range(1, 10))
assert cursor.rowcount == 9
assert cursor.rownumber == 8
assert cursor.fetchone() is None

# Verify all column names are present (case insensitive)
col_names = [col.column_name.lower() for col in cols]
Expand Down Expand Up @@ -14055,7 +14059,24 @@ def test_columns_table_pattern(cursor):
"""Test columns with table name pattern"""
try:
# Get columns with table pattern
arraysize = cursor.arraysize
cols = cursor.columns(table="columns_%", schema="pytest_cols_schema").fetchall()
description = cursor.description
assert len(cols) == 18
assert cursor.rowcount == 18
assert cursor.rownumber == 17
assert cursor.fetchone() is None
assert cursor.arraysize == arraysize

# Compare all ordered cells and types with the independent row-wise fetch path.
cursor.columns(table="columns_%", schema="pytest_cols_schema")
assert type(cursor.description) is type(description)
assert cursor.description == description
expected = [tuple(row) for row in cursor]
assert [tuple(row) for row in cols] == expected
assert [[type(value) for value in row] for row in cols] == [
[type(value) for value in row] for row in expected
]

# Should find columns from both test tables
tables_found = set()
Expand All @@ -14068,6 +14089,33 @@ def test_columns_table_pattern(cursor):
"columns_special_test" in tables_found
), "Should find columns_special_test with pattern columns_%"

# Cross both catalog batch-growth boundaries without creating another table.
extra_columns = [f"fetch_growth_{i:03d}" for i in range(100)]
cursor.execute(
"ALTER TABLE pytest_cols_schema.columns_special_test ADD "
+ ", ".join(f"[{name}] INT NULL" for name in extra_columns)
)
try:
expanded = cursor.columns(table="columns_%", schema="pytest_cols_schema").fetchall()
assert cursor.description == description
assert len(expanded) == 118
assert cursor.rowcount == 118
assert cursor.rownumber == 117
assert cursor.fetchone() is None
assert cursor.arraysize == arraysize

cursor.columns(table="columns_%", schema="pytest_cols_schema")
expected = [tuple(row) for row in cursor]
assert [tuple(row) for row in expanded] == expected
assert [[type(value) for value in row] for row in expanded] == [
[type(value) for value in row] for row in expected
]
finally:
cursor.execute(
"ALTER TABLE pytest_cols_schema.columns_special_test DROP COLUMN "
+ ", ".join(f"[{name}]" for name in extra_columns)
)

finally:
# Clean up happens in test_columns_cleanup
pass
Expand Down Expand Up @@ -17649,6 +17697,26 @@ def test_columns_fetchone(cursor, db_connection, catalog_fetch_schema):
assert row is not None, "fetchone() should return a row from columns()"
assert hasattr(row, "column_name")
assert row.table_name.lower() == "fetch_test"
rows = [row] + cursor.fetchmany(1) + cursor.fetchall()
assert [item.column_name for item in rows] == ["id", "name", "value", "ts"]
assert [item.ordinal_position for item in rows] == [1, 2, 3, 4]
assert cursor.rowcount == 4
assert cursor.rownumber == 3
assert cursor.fetchone() is None
statement = cursor.hstmt
operation = "SELECT ? AS ordinary; SELECT ? AS next_result"
cursor.execute(operation, (1, 2))
assert cursor.hstmt is statement
assert cursor.fetchall()[0].ordinary == 1
assert cursor.nextset() is True
assert cursor.fetchall()[0].next_result == 2
assert cursor.nextset() is False
cursor.execute(operation, (3, 4), reset_cursor=False)
assert cursor.hstmt is statement
assert cursor.fetchall()[0].ordinary == 3
assert cursor.nextset() is True
assert cursor.fetchall()[0].next_result == 4
assert cursor.nextset() is False


def test_primarykeys_fetchone(cursor, db_connection, catalog_fetch_schema):
Expand Down
Loading
Loading