Simplify code comments added since 1.0a40 (#2957)

Refs #2867
This commit is contained in:
Simon Willison 2026-09-24 13:52:29 -07:00 • committed by GitHub
commit 8e17729ff3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
14 changed files with 412 additions and 1451 deletions

View file

@ -26,8 +26,8 @@ SECRET_PARAM_VALUE = "SUPER_SECRET_PARAM_VALUE_XYZ_123"
INVALID_SQL = "select this_is_not_valid_sql from nowhere"
# Bounded so a broken time limit fails the test instead of hanging it, but far
# too long to finish inside any of the millisecond budgets used below.
# Bounded so a broken time limit fails rather than hangs, but too slow to
# finish within the millisecond time limits used below.
SLOW_SQL = """
with recursive counter(x) as (
select 1 union all select x + 1 from counter where x < 50000000
@ -41,13 +41,7 @@ def _db_query_spans(otel_spans):
def _spans_for_namespace(otel_spans, namespace):
"""
db.query spans belonging to one database.
Datasette queries its internal catalog constantly - including while a
Datasette instance is being constructed - so a test that just grabbed
every db.query span would be reading someone else's traffic.
"""
"db.query spans for one database, excluding queries against the internal database."
return [
span
for span in _db_query_spans(otel_spans)
@ -56,14 +50,7 @@ def _spans_for_namespace(otel_spans, namespace):
def _children_named(otel_spans, name, parent_span_context):
"""
Finished spans called `name` whose parent really is `parent_span_context`.
Parentage is matched on span id, not on "a span with this name exists" -
a span can exist and still be an unparented root if a thread boundary
dropped the otel context, which is the exact failure these tests exist
to catch.
"""
"Finished spans called `name` that are direct children of `parent_span_context`."
return [
span
for span in otel_spans.get_finished_spans()
@ -76,12 +63,7 @@ def _children_named(otel_spans, name, parent_span_context):
def _descends_from(span, ancestor_span_context, by_span_id):
"""
True if `span` reaches `ancestor_span_context` by walking parent links.
Walks real span ids rather than trusting a shared trace id: a span can
carry the right trace id and still hang off the wrong parent.
"""
"True if `span` reaches `ancestor_span_context` by walking parent links."
seen = set()
current = span
while current.parent is not None:
@ -97,7 +79,7 @@ def _descends_from(span, ancestor_span_context, by_span_id):
def _all_attribute_values(otel_spans):
"Every attribute value across every finished span, for the 'no leaked param values' test."
"Every attribute value on every finished span and span event."
values = []
for span in otel_spans.get_finished_spans():
values.extend((span.attributes or {}).values())
@ -108,14 +90,9 @@ def _all_attribute_values(otel_spans):
def test_datasette_package_never_imports_the_sdk():
"""
Core depends on opentelemetry-api only. The SDK is a test dependency.
Importing datasette does not load the OpenTelemetry SDK.
Checked by importing datasette in a fresh process and inspecting
sys.modules, rather than by grepping, so a lazy `import
opentelemetry.sdk` inside a function body cannot slip past.
conftest.py's pytest_collection_modifyitems() moves this test to the
front of the run by name - if you rename it, rename it there too.
conftest.py moves this test to the front of the run by name.
"""
code = (
"import datasette.app, datasette.database, datasette.telemetry, sys; "
@ -149,13 +126,7 @@ async def test_db_query_span_basic_attributes(ds_client, otel_spans):
@pytest.mark.asyncio
async def test_truncated_result_sets_truncated_attribute(otel_spans):
"""
A result actually cut short by max_returned_rows records truncated=True.
Every other test asserts the attribute is False, so a regression that
recorded the flag before the slice (or inverted it) would pass the rest
of the suite.
"""
"A result cut short by max_returned_rows records truncated=True."
ds = Datasette(memory=True, settings={"max_returned_rows": 5})
db = ds.add_memory_database("t04_truncated")
results = await db.execute(
@ -178,8 +149,7 @@ async def test_facetable_request_produces_db_query_spans(ds_client, otel_spans):
spans = _db_query_spans(otel_spans)
assert spans, "expected at least one db.query span"
assert all(span.attributes["db.system"] == "sqlite" for span in spans)
# Every db.query names what ran: SQL text for the string methods,
# datasette.callback for callback-style calls (schema introspection here).
# Each span records the SQL or, for callback methods, the callback name:
assert all(
span.attributes.get("db.query.text")
or span.attributes.get("datasette.callback")
@ -194,8 +164,7 @@ async def test_facetable_request_produces_db_query_spans(ds_client, otel_spans):
def test_sql_attribute_truncates_at_2048():
short_sql = "select 1"
assert sql_attribute(short_sql) == "select 1"
# Whitespace is stripped, so the same query logged twice with different
# surrounding whitespace produces one attribute value, not two.
# Surrounding whitespace is stripped:
assert sql_attribute(" select 1\n") == "select 1"
long_sql = "select 1 -- " + ("x" * 3000)
@ -207,8 +176,7 @@ def test_sql_attribute_truncates_at_2048():
@pytest.mark.asyncio
async def test_db_query_text_is_truncated_in_real_span(ds_client, otel_spans):
# A long trailing SQL comment keeps the query valid and executable while
# pushing db.query.text well past the 2048 char cap.
# A long trailing comment keeps the SQL valid but over the 2048 character limit
long_sql = "select 1 -- " + ("x" * 3000)
response = await ds_client.get("/fixtures/-/query.json", params={"sql": long_sql})
assert response.status_code == 200
@ -231,8 +199,7 @@ async def test_no_span_attribute_ever_contains_a_parameter_value(ds_client, otel
params={"sql": "select :secret", "secret": SECRET_PARAM_VALUE},
)
assert response.status_code == 200
# Sanity check the value really did flow through as a bound parameter,
# not inlined into the SQL text, otherwise this test would be vacuous.
# Confirm the bound parameter value was used by the query:
assert SECRET_PARAM_VALUE in json.dumps(response.json())
for value in _all_attribute_values(otel_spans):
@ -253,13 +220,10 @@ async def test_no_span_attribute_ever_contains_a_parameter_value(ds_client, otel
@pytest.mark.asyncio
async def test_query_interrupted_sets_error_status(otel_spans):
"""
A query that runs out the instance-wide sql_time_limit_ms is an error.
A query that exceeds the sql_time_limit_ms setting is a span error.
This used to force the timeout with `?_timelimit=5`, but a caller-supplied
budget shorter than the instance limit is now the signal that the timeout
was expected - see test_expected_timeout_is_not_a_span_error - so the
timeout has to come from the setting for this to still test what it was
written to test.
The limit comes from the setting because a shorter custom_time_limit
marks the timeout as expected.
"""
ds = Datasette(memory=True, settings={"sql_time_limit_ms": 20})
db = ds.add_memory_database("t09_instance_limit_timeout")
@ -276,23 +240,14 @@ async def test_query_interrupted_sets_error_status(otel_spans):
async def _expected_timeout_count_span(otel_spans, database_name):
"""
Drive the real table_counts() path into a timeout; return its db.query span.
table_counts() is where the headline instance of this lives: the homepage
counts every table under a 10ms budget and stores None for any table that
does not finish in time. Before this was fixed, a two-table database
produced four ERROR spans - two db.query and two db.query.execute - on
every single homepage hit.
"""
"Make table_counts() time out and return its db.query span."
db = Datasette(memory=True).add_memory_database(database_name)
await db.execute_write("create table big (id integer primary key, t text)")
await db.execute_write_many(
"insert into big (t) values (?)", [["x" * 50] for _ in range(11000)]
)
# count_limit caps the scan at 10001 rows, and below 20ms sqlite_timelimit()
# runs its progress handler on every VM instruction, so 1ms is not a close
# call - a scan of that size takes single-digit milliseconds at best.
# count_limit caps the scan at 10001 rows. Below 20ms sqlite_timelimit()
# checks the limit on every VM instruction, so this reliably exceeds 1ms.
counts = await db.table_counts(1)
assert counts == {
"big": None
@ -310,7 +265,7 @@ async def _expected_timeout_count_span(otel_spans, database_name):
@pytest.mark.asyncio
async def test_expected_timeout_is_not_a_span_error(otel_spans):
span = await _expected_timeout_count_span(otel_spans, "t09_expected_timeout")
# The useful signal survives; only the red status goes away.
# Recorded as interrupted, but not as an error:
assert span.attributes["datasette.interrupted"] is True
assert span.status.status_code != StatusCode.ERROR
assert not [event for event in span.events if event.name == "exception"]
@ -318,13 +273,7 @@ async def test_expected_timeout_is_not_a_span_error(otel_spans):
@pytest.mark.asyncio
async def test_expected_timeout_does_not_error_the_inner_execute_span(otel_spans):
"""
The same fix has to reach db.query.execute, which sets its own status.
Half of the original bug lived here: the inner span passed
set_status_on_exception=log_sql_errors, and table_counts() leaves
log_sql_errors at its True default, so it went ERROR too.
"""
"The db.query.execute child span is not marked as an error either."
span = await _expected_timeout_count_span(otel_spans, "t09_expected_timeout_inner")
children = _children_named(otel_spans, "db.query.execute", span.context)
assert len(children) == 1
@ -335,13 +284,7 @@ async def test_expected_timeout_does_not_error_the_inner_execute_span(otel_spans
@pytest.mark.asyncio
async def test_unexpected_timeout_is_still_a_span_error(otel_spans):
"""
A custom_time_limit *above* sql_time_limit_ms is not a short budget.
This is the half of the rule that stops the fix collapsing into "never
report timeouts": the caller asked for 5 seconds, the instance overruled it
at 20ms, and nobody expected that.
"""
"A timeout is an error if custom_time_limit is above sql_time_limit_ms."
ds = Datasette(memory=True, settings={"sql_time_limit_ms": 20})
db = ds.add_memory_database("t09_custom_limit_ignored")
with pytest.raises(QueryInterrupted):
@ -350,8 +293,7 @@ async def test_unexpected_timeout_is_still_a_span_error(otel_spans):
spans = _spans_for_namespace(otel_spans, "t09_custom_limit_ignored")
assert spans
span = spans[-1]
# Proves the caller's larger budget really was discarded - otherwise this
# would be asserting on a query that ran under a 5s limit.
# The setting overrides the larger custom_time_limit:
assert span.attributes["datasette.time_limit_ms"] == 20
assert span.attributes["datasette.interrupted"] is True
assert span.status.status_code == StatusCode.ERROR
@ -378,14 +320,7 @@ async def test_unsuppressed_sql_error_is_a_span_error(ds_client, otel_spans):
@pytest.mark.asyncio
async def test_suppressed_sql_error_is_not_a_span_error(ds_client, otel_spans):
"""
log_sql_errors=False means the caller is probing and expects failures.
Facet suggestion runs `json_type(column)` against every column precisely
to discover which ones raise, so marking those spans as errors would put
two red spans per text column on every table page - burying real failures
and tripping any alerting keyed on span status.
"""
"With log_sql_errors=False the error is recorded as suppressed, not a span error."
db = ds_client.ds.get_database("fixtures")
with pytest.raises(sqlite3.OperationalError):
await db.execute(INVALID_SQL, log_sql_errors=False)
@ -400,8 +335,7 @@ async def test_suppressed_sql_error_is_not_a_span_error(ds_client, otel_spans):
@pytest.mark.asyncio
async def test_execute_write_produces_db_query_span(otel_spans):
# Named in-memory databases are shared-cache, so every test in this file
# needs its own name or the second `create table` hits an existing table.
# Named in-memory databases are shared, so each test uses a unique name.
db = Datasette(memory=True).add_memory_database("t03_write_span")
await db.execute_write("create table docs (id integer primary key, name text)")
await db.execute_write("insert into docs (id, name) values (?, ?)", [1, "one"])
@ -451,27 +385,18 @@ async def test_execute_write_many_records_param_sets_not_rows_returned(otel_span
span = many_spans[0]
assert span.attributes["datasette.param_sets"] == 5
# executemany() consumes parameter sets and returns no rows at all, so
# calling this a row count would be a lie. Asserted explicitly because the
# attribute really was named datasette.rows_returned at one point.
assert "datasette.rows_returned" not in span.attributes
# --- Context propagation across thread boundaries --------------------------
#
# Every assertion below checks parentage (child.parent.span_id ==
# expected_parent.span_id, in the same trace), not merely that spans exist.
# Spans can exist and still be wrongly parented - or be unparented roots - if
# a thread boundary drops the otel context, which is exactly the failure mode
# these tests exist to prevent.
# These tests check span parentage, not just that the spans exist.
@pytest.mark.asyncio
async def test_db_query_execute_parents_to_db_query(ds_client, otel_spans):
# execute_fn()'s executor.submit() is thread boundary #1. The
# db.query.execute span is created inside the worker thread; without the
# copy_context() propagation it comes back as an unparented root span
# rather than a child of db.query.
# execute_fn() submits to the executor, so db.query.execute is created on
# another thread.
response = await ds_client.get("/fixtures/-/query.json?sql=select+1")
assert response.status_code == 200
@ -490,19 +415,15 @@ async def test_db_query_execute_parents_to_db_query(ds_client, otel_spans):
], "expected at least one db.query.execute span"
children = _children_named(otel_spans, "db.query.execute", query_span.context)
assert len(children) == 1, "expected exactly one db.query.execute child of db.query"
# The execute span is strictly contained by the round-trip span, and the
# gap between the two is the thread-pool wait.
# db.query.execute runs within db.query; the gap is the thread pool wait.
assert query_span.start_time <= children[0].start_time
assert children[0].end_time <= query_span.end_time
@pytest.mark.asyncio
async def test_immutable_database_propagates_context(tmp_path, otel_spans):
# Thread boundary #3, the easy one to miss: immutable databases route
# execute_isolated_fn() through loop.run_in_executor() directly rather
# than through the write thread. A span created inside that worker must
# still parent to whatever was current when execute_isolated_fn() was
# awaited, or every immutable-database operation emits orphan roots.
# Immutable databases run execute_isolated_fn() on another thread using
# loop.run_in_executor(), not the write thread.
db_path = tmp_path / "t04_immutable.db"
sqlite_utils.Database(str(db_path))["t"].insert({"id": 1}, pk="id")
@ -526,10 +447,7 @@ async def test_immutable_database_propagates_context(tmp_path, otel_spans):
for span in otel_spans.get_finished_spans()
if span.name == "t04-child-in-isolated-worker"
], "expected a span created inside execute_isolated_fn's worker thread"
# execute_isolated_fn() now opens its own db.query span, so the chain is
# event-loop parent -> db.query -> worker child. The worker child
# parenting to that db.query span, across the thread, is the propagation
# this test exists to prove.
# Expected chain: event loop parent -> db.query -> worker thread child
query_spans = _children_named(otel_spans, "db.query", parent_context)
assert len(query_spans) == 1
children = _children_named(
@ -540,10 +458,8 @@ async def test_immutable_database_propagates_context(tmp_path, otel_spans):
@pytest.mark.asyncio
async def test_write_spans_parent_to_db_query(otel_spans):
# Thread boundary #2: WriteTask -> queue.Queue -> the write thread.
# db.write.queue_wait and db.write.execute are both direct children of
# the db.query span that was current on the event loop at enqueue time,
# so they are siblings rather than nested inside one another.
# execute_write() queues a WriteTask for the write thread.
# db.write.queue_wait and db.write.execute are both children of db.query.
db = Datasette(memory=True).add_memory_database("t04_write_spans")
await db.execute_write("create table docs (id integer primary key)")
@ -563,18 +479,14 @@ async def test_write_spans_parent_to_db_query(otel_spans):
execute_span = execute_children[0]
assert execute_span.attributes["datasette.isolated_connection"] is False
assert execute_span.attributes["datasette.transaction"] is True
# Siblings, not parent/child: the queue wait is over by the time the
# write begins.
# The queue wait ends before the write begins.
assert queue_wait_children[0].end_time <= execute_span.start_time
@pytest.mark.asyncio
async def test_write_queue_wait_duration_reflects_real_wait(otel_spans):
# db.write.queue_wait is built from explicit start/end timestamps -
# task.enqueued_at_ns, captured on the event loop, through to the moment
# the write thread dequeued it. If it were a plain `with` block on the
# write thread it would instead measure the microseconds spent building
# the span object, and this assertion would fail.
# db.write.queue_wait runs from task.enqueued_at_ns, captured on the event
# loop, to when the write thread dequeues the task.
ds = Datasette(memory=True)
db = ds.add_memory_database("t04_queue_wait")
await db.execute_write("create table docs (id integer primary key)")
@ -582,9 +494,7 @@ async def test_write_queue_wait_duration_reflects_real_wait(otel_spans):
def slow_write(conn):
time.sleep(0.1)
# Queue a deliberately slow write without waiting for it, then queue a
# second write immediately behind it: the second task sits in the queue
# for roughly the duration of the first.
# Queue a slow write without waiting for it, then a second write behind it:
_, slow_future = await db._send_to_write_thread(slow_write, block=False)
await db.execute_write("insert into docs (id) values (1)")
await slow_future
@ -600,23 +510,17 @@ async def test_write_queue_wait_duration_reflects_real_wait(otel_spans):
)
assert len(queue_wait_children) == 1
duration_ns = queue_wait_children[0].end_time - queue_wait_children[0].start_time
# The slow write sleeps 100ms; anything above 10ms is far beyond the
# microseconds a mis-timestamped span would report.
# The slow write sleeps for 100ms
assert duration_ns > 10_000_000, f"queue wait was only {duration_ns}ns"
async def _write_spans_from_one_enqueue(otel_spans, name, block):
"""
Run exactly one write through the write thread from inside a span of our
own, and return (enqueueing span context, {span name: span}).
Run one write through the write thread inside a span, returning
(enqueueing span context, {span name: span}).
`_send_to_write_thread` is called directly rather than `execute_write()`
because `execute_write()` opens its own db.query span, which would then
be the span current at enqueue time - so the parent/link would point at
that span rather than at the one this test controls.
The exporter is cleared immediately before the enqueue so the write spans
collected here can only have come from this one write.
Uses _send_to_write_thread() because execute_write() would add its own
db.query span between the enqueueing span and the write spans.
"""
db = Datasette(memory=True).add_memory_database(name)
await db.execute_write("create table docs (id integer primary key)")
@ -629,11 +533,8 @@ async def _write_spans_from_one_enqueue(otel_spans, name, block):
enqueuer_context = enqueuer.get_span_context()
queued = await db._send_to_write_thread(insert, block=block)
if not block:
# The point of block=False is that the write happens after the
# caller has returned and the enqueueing span above has closed.
# Awaiting the reply future outside that `with` waits for the write
# thread deterministically - it is resolved only after both write
# spans have ended and been exported.
# Wait for the write after the enqueueing span has ended. The reply
# future resolves once both write spans have been exported.
_, reply_future = queued
await reply_future
@ -648,9 +549,8 @@ async def _write_spans_from_one_enqueue(otel_spans, name, block):
@pytest.mark.asyncio
async def test_blocking_write_spans_still_parent_normally(otel_spans):
# Regression guard for ticket 07: block=True genuinely has containment -
# the caller awaits the reply future - so those spans must keep parenting
# to the enqueueing span, and must not grow links.
# block=True waits for the write, so its spans are children of the
# enqueueing span, with no links.
enqueuer_context, spans = await _write_spans_from_one_enqueue(
otel_spans, "t07_blocking_write", block=True
)
@ -664,24 +564,21 @@ async def test_blocking_write_spans_still_parent_normally(otel_spans):
@pytest.mark.asyncio
async def test_nonblocking_write_spans_are_roots_with_a_link(otel_spans):
# block=False returns before the write runs, so the enqueueing span has
# already ended (and exported) by the time these spans start. Parenting
# them to it would draw a child outliving its closed parent, so they are
# roots in their own traces, linked back to the span that caused them.
# block=False returns before the write runs, so the write spans are roots
# linked to the enqueueing span.
enqueuer_context, spans = await _write_spans_from_one_enqueue(
otel_spans, "t07_nonblocking_write", block=False
)
assert enqueuer_context.is_valid, "test's own enqueueing span was not recorded"
for name, span in spans.items():
assert span.parent is None, f"{name} is still parented"
# A link does not join the linked trace: each of these is its own
# root trace, which is the correct shape and not a workaround.
# Each write span starts its own trace
assert span.context.trace_id != enqueuer_context.trace_id, name
assert len(span.links) == 1, f"{name} has links {span.links}"
link_context = span.links[0].context
assert link_context.trace_id == enqueuer_context.trace_id, name
assert link_context.span_id == enqueuer_context.span_id, name
# The two write spans are independent roots, not nested in one another.
# The two write spans are separate roots
assert (
spans["db.write.queue_wait"].context.trace_id
!= spans["db.write.execute"].context.trace_id
@ -690,8 +587,6 @@ async def test_nonblocking_write_spans_are_roots_with_a_link(otel_spans):
@pytest.mark.asyncio
async def test_nonblocking_write_link_has_no_attributes(otel_spans):
# There is only one kind of link here, so a relationship-name attribute
# would be a constant conveying nothing the link's existence does not.
_, spans = await _write_spans_from_one_enqueue(
otel_spans, "t07_nonblocking_link_attrs", block=False
)
@ -705,18 +600,10 @@ async def test_nonblocking_write_spans_ignore_the_write_threads_ambient_context(
otel_spans,
):
"""
block=False spans pass an explicit empty Context, not merely "no attach".
block=False spans ignore any context left attached on the write thread.
Nothing is attached for a block=False task, but "nothing attached" is not
the same as "no ambient context": the write thread is persistent, and
anything running on it - a prepare_connection plugin hook, say - can
attach a context and never detach it. Without the explicit `context=`
these spans would silently parent to that leftover span instead of being
roots, and no other test here would notice, because in every other test
the write thread's ambient context happens to be empty.
So this test leaks exactly such a context on the write thread, the way a
careless plugin would, and then checks the write spans are still roots.
A prepare_connection hook could attach a context and never detach it.
This test does that, then checks the write spans are still roots.
"""
ds = Datasette(memory=True)
db = ds.add_memory_database("t07_ambient_write_thread")
@ -726,8 +613,8 @@ async def test_nonblocking_write_spans_ignore_the_write_threads_ambient_context(
def prepare_connection(conn, database):
if threading.current_thread().name == write_thread_name:
# Runs once, on the write thread, before any task is dequeued -
# and never detaches, which is the whole point.
# Runs on the write thread before any task is dequeued, and never
# detaches.
span = tracer.start_span("leaked-write-thread-ambient-span")
leaked["span_id"] = span.get_span_context().span_id
otel_context_api.attach(otel_trace.set_span_in_context(span))
@ -766,14 +653,7 @@ async def test_nonblocking_write_spans_ignore_the_write_threads_ambient_context(
@pytest.mark.asyncio
async def test_suppressed_error_does_not_mark_execute_span(ds_client, otel_spans):
"""
The inner db.query.execute span must honour log_sql_errors too.
It is created inside the worker thread, so without record_exception /
set_status_on_exception being passed through it would mark every facet
suggestion probe as failed even though the outer db.query span correctly
reports the failure as suppressed.
"""
"The inner db.query.execute span also respects log_sql_errors=False."
db = ds_client.ds.get_database("fixtures")
with pytest.raises(sqlite3.OperationalError):
await db.execute(INVALID_SQL, log_sql_errors=False)
@ -791,23 +671,14 @@ async def test_suppressed_error_does_not_mark_execute_span(ds_client, otel_spans
@pytest.mark.asyncio
async def test_invoke_startup_produces_one_trace_not_dozens_of_orphans(otel_spans):
"""
invoke_startup() runs with no request, so nothing it does has an ambient
span to nest under. Without datasette.startup every register_* hook, every
internal-catalog read and every catalog write becomes its own single-span
root trace - around twenty of them per fresh instance.
"""
"Spans emitted by invoke_startup() share a single datasette.startup root span."
ds = Datasette(memory=True)
# Named in-memory databases are shared-cache, so this needs its own name.
ds.add_memory_database("t05_startup_db")
# Constructing a Datasette already touches the internal catalog, and that
# work is genuinely outside startup. Clear so the assertions below describe
# invoke_startup() alone.
# Ignore spans from constructing Datasette, which happens before startup
otel_spans.clear()
# Deliberately no ambient span: this mirrors the ASGI lifespan path, where
# startup runs before any request exists. If something did wrap this call
# the "one root" assertion below would pass for the wrong reason.
# No ambient span, as in the ASGI lifespan path where startup runs before
# any request.
assert (
not otel_trace.get_current_span().get_span_context().is_valid
), "this test must run with no ambient span"
@ -833,7 +704,7 @@ async def test_invoke_startup_produces_one_trace_not_dozens_of_orphans(otel_span
by_span_id = {span.context.span_id: span for span in spans}
# The internal catalog reads are what made up the bulk of the orphans.
# Internal database reads:
internal_queries = [
span
for span in spans
@ -844,8 +715,7 @@ async def test_invoke_startup_produces_one_trace_not_dozens_of_orphans(otel_span
_descends_from(span, startup.context, by_span_id) for span in internal_queries
)
# ...and the catalog writes, which reach the span through the write thread,
# so they also prove the ticket-04 context capture survives startup.
# Internal database writes, which run on the write thread:
write_spans = [span for span in spans if span.name.startswith("db.write.")]
assert write_spans, "expected db.write.* spans during startup"
assert all(
@ -859,20 +729,11 @@ async def test_invoke_startup_produces_one_trace_not_dozens_of_orphans(otel_span
@pytest.mark.asyncio
async def test_db_query_is_client_kind_and_children_are_internal(otel_spans):
"""
db.query is a database client span; Datasette's decomposition of it is not.
Trace UIs key their database rendering off the span kind rather than off
db.system, so db.query has to be CLIENT. db.query.execute,
db.write.execute and db.write.queue_wait deliberately stay INTERNAL: they
are parts of one logical query rather than three separate database calls,
and queue_wait touches no database at all - marking them CLIENT would
make one query look like several to anything counting spans by kind.
db.query spans are CLIENT. Their child spans are INTERNAL because they are
parts of one query rather than separate database calls.
"""
# Named in-memory databases are shared-cache, so this needs its own name.
db = Datasette(memory=True).add_memory_database("t06_span_kind")
# All four db.query entry points, so a missed `kind=` on any one of them
# fails here - plus the write path (db.write.queue_wait,
# db.write.execute) and the read path (db.query.execute) children.
# Call each of the four SQL string methods:
await db.execute_write("create table docs (id integer primary key)")
await db.execute_write_many(
"insert into docs (id) values (?)", [[i] for i in range(1, 4)]
@ -899,15 +760,7 @@ async def test_db_query_is_client_kind_and_children_are_internal(otel_spans):
async def test_instrumentation_scope_declares_version_and_schema_url(
ds_client, otel_spans
):
"""
Spans say which Datasette produced them and which semconv version their
attribute names follow.
Before get_tracer() was given a version and a schema URL every exported
scope was name='datasette' version='' schema_url='', so nothing
downstream could tell which Datasette a span came from, or whether
`db.system` meant `db.system` or the post-1.30.0 `db.system.name`.
"""
"The instrumentation scope includes the Datasette version and schema URL."
response = await ds_client.get("/fixtures/-/query.json?sql=select+1")
assert response.status_code == 200
@ -917,10 +770,7 @@ async def test_instrumentation_scope_declares_version_and_schema_url(
assert scope.name == "datasette"
assert scope.version == __version__
# The literal URL, not the SCHEMA_URL constant: comparing the span
# against the same constant the instrumentation is built from would only
# catch a dropped argument, never a wrong value. Bumping this is a claim
# about the attribute names on the wire - see SCHEMA_URL in telemetry.py.
# Uses the literal URL so changing SCHEMA_URL requires updating this test
assert scope.schema_url == "https://opentelemetry.io/schemas/1.29.0"
assert SCHEMA_URL == "https://opentelemetry.io/schemas/1.29.0"
assert __version__, "the scope version must not be empty"
@ -929,14 +779,11 @@ async def test_instrumentation_scope_declares_version_and_schema_url(
def test_db_operation_name_from_leading_keyword():
assert sql_operation_name("select 1") == "SELECT"
assert sql_operation_name(" insert into x (a) values (1)") == "INSERT"
# A leading CTE reports WITH rather than the operation inside it. That is
# the documented limitation, not an accident - see sql_operation_name().
# A leading CTE reports WITH, not the operation inside it
assert sql_operation_name("with foo as (select 1) select * from foo") == "WITH"
# Unrecognised leading keyword: no attribute rather than a wrong one, and
# no unbounded value set derived from attacker-supplied SQL.
# Unrecognized leading keyword
assert sql_operation_name("gibberish 1") is None
# Not a parser: a parenthesised SELECT and a leading comment both yield
# nothing rather than a guess.
# A parenthesized SELECT or a leading comment also returns None
assert sql_operation_name("(select 1) union select 2") is None
assert sql_operation_name("-- a comment\nselect 1") is None
assert sql_operation_name("") is None
@ -976,14 +823,10 @@ async def test_execute_write_sets_db_operation_name(otel_spans):
@pytest.mark.asyncio
async def test_execute_write_script_has_no_operation_name(otel_spans):
"""
executescript() runs several statements, so naming the operation after
the first one would be a lie. Semantic conventions say db.operation.name
should not be extracted from query text that can hold more than one
operation, so the attribute is absent entirely.
Scripts can contain several statements, so db.operation.name is omitted.
The script deliberately starts with `create`, which *is* on the
allowlist - so this fails if the call site ever starts calling
sql_operation_name().
The script starts with `create`, which is on the allowlist, so this fails
if the operation name is extracted anyway.
"""
db = Datasette(memory=True).add_memory_database("t06_script_operation")
await db.execute_write_script(
@ -1022,8 +865,7 @@ async def test_execute_fn_produces_db_query_span(otel_spans):
span.attributes["datasette.callback"]
== "test_execute_fn_produces_db_query_span.<locals>.count_rows"
)
# There is no SQL string for a callback, and no statement to take a
# leading keyword from - absent beats guessed.
# Callbacks have no SQL text to record or take an operation name from
assert "db.query.text" not in span.attributes
assert "db.operation.name" not in span.attributes
children = _children_named(otel_spans, "db.query.execute", span.context)
@ -1032,7 +874,6 @@ async def test_execute_fn_produces_db_query_span(otel_spans):
@pytest.mark.asyncio
async def test_execute_fn_lambda_reports_lambda(otel_spans):
# Pins the documented behaviour rather than pretending lambdas have names.
db = Datasette(memory=True).add_memory_database("t16_lambda")
otel_spans.clear()
await db.execute_fn(lambda conn: conn.execute("select 1").fetchone())
@ -1067,9 +908,7 @@ async def test_execute_write_fn_produces_db_query_span(otel_spans):
@pytest.mark.asyncio
async def test_execute_write_fn_callback_name_is_not_the_hook_wrapper(otel_spans):
# A callback that declares track_event is the case where
# _wrap_fn_with_hooks() actually replaces fn with a wrapper - the span
# must still report the caller's function, not the wrapper's name.
# _wrap_fn_with_hooks() wraps callbacks that accept track_event
db = Datasette(memory=True).add_memory_database("t16_wrapper_name")
def create_with_events(conn, track_event):
@ -1087,9 +926,8 @@ async def test_execute_write_fn_callback_name_is_not_the_hook_wrapper(otel_spans
@pytest.mark.asyncio
async def test_execute_write_fn_nonblocking_spans_link_to_the_new_span(otel_spans):
# For block=False the public db.query span ends at enqueue and the
# write-thread spans become roots. Their link must target that new span,
# not whatever was current around the execute_write_fn() call.
# With block=False the write thread spans link to the db.query span from
# execute_write_fn(), not to the span that was current when it was called.
db = Datasette(memory=True).add_memory_database("t16_nonblocking")
await db.execute_write("create table docs (id integer primary key)")
@ -1100,8 +938,7 @@ async def test_execute_write_fn_nonblocking_spans_link_to_the_new_span(otel_span
with tracer.start_as_current_span("t16-enqueueing-span") as enqueuer:
enqueuer_context = enqueuer.get_span_context()
await db.execute_write_fn(insert, block=False)
# Writes are serialized on the write thread, so a blocking write behind
# the non-blocking one waits for it deterministically.
# Writes run in order, so this waits for the non-blocking write to finish
await db.execute_write("insert into docs (id) values (2)")
query_spans = [
@ -1125,9 +962,8 @@ async def test_execute_write_fn_nonblocking_spans_link_to_the_new_span(otel_span
@pytest.mark.asyncio
async def test_execute_does_not_double_wrap(otel_spans):
# The regression guard for the refactor: execute() and the SQL-string
# write methods call the private _execute_fn/_execute_write_fn, so they
# must not gain a second db.query span from the public wrappers.
# execute() and the SQL string write methods call the private
# _execute_fn() and _execute_write_fn(), so they create one db.query span.
db = Datasette(memory=True).add_memory_database("t16_no_double_wrap")
otel_spans.clear()
await db.execute_write("create table t (id integer primary key)")
@ -1172,8 +1008,7 @@ async def test_execute_isolated_fn_span_on_mutable_and_immutable(tmp_path, otel_
@pytest.mark.asyncio
async def test_execute_fn_exception_marks_span_error(otel_spans):
# Unlike execute(), there is no probing caller on this path - a callback
# that raises is an error, with the default record_exception behaviour.
# execute_fn() has no log_sql_errors option, so exceptions are span errors
db = Datasette(memory=True).add_memory_database("t16_fn_error")
def boom(conn):