This commit is contained in:
lbellomo 2024-09-20 18:24:03 -04:00 committed by GitHub
commit 48bd369f44
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 25 additions and 11 deletions

View file

@ -1,7 +1,7 @@
import base64
import click
from click_default_group import DefaultGroup # type: ignore
from datetime import datetime
from datetime import datetime, timezone
import hashlib
import pathlib
from runpy import run_module
@ -865,7 +865,7 @@ def insert_upsert_options(*, require_pk=False):
required=True,
),
click.argument("table"),
click.argument("file", type=click.File("rb"), required=True),
click.argument("file", type=click.File("rb", lazy=True), required=True),
click.option(
"--pk",
help="Columns to use as the primary key, e.g. id",
@ -1949,12 +1949,16 @@ def memory(
fp = file_path.open("rb")
rows, format_used = rows_from_file(fp, format=format, encoding=encoding)
tracker = None
if format_used in (Format.CSV, Format.TSV) and not no_detect_types:
tracker = TypeTracker()
rows = tracker.wrap(rows)
if flatten:
rows = (_flatten(row) for row in rows)
db[file_table].insert_all(rows, alter=True)
try:
if format_used in (Format.CSV, Format.TSV) and not no_detect_types:
tracker = TypeTracker()
rows = tracker.wrap(rows)
if flatten:
rows = (_flatten(row) for row in rows)
db[file_table].insert_all(rows, alter=True)
except UnicodeDecodeError as e:
fp.close()
raise e
if tracker is not None:
db[file_table].transform(types=tracker.types)
# Add convenient t / t1 / t2 views
@ -3196,8 +3200,12 @@ FILE_COLUMNS = {
"ctime": lambda p: p.stat().st_ctime,
"mtime_int": lambda p: int(p.stat().st_mtime),
"ctime_int": lambda p: int(p.stat().st_ctime),
"mtime_iso": lambda p: datetime.utcfromtimestamp(p.stat().st_mtime).isoformat(),
"ctime_iso": lambda p: datetime.utcfromtimestamp(p.stat().st_ctime).isoformat(),
"mtime_iso": lambda p: datetime.fromtimestamp(p.stat().st_mtime, timezone.utc)
.isoformat()
.rstrip("+00:00"),
"ctime_iso": lambda p: datetime.fromtimestamp(p.stat().st_mtime, timezone.utc)
.isoformat()
.rstrip("+00:00"),
"size": lambda p: p.stat().st_size,
"stem": lambda p: p.stem,
"suffix": lambda p: p.suffix,

View file

@ -212,6 +212,7 @@ def _extra_key_strategy(
reader: Iterable[dict],
ignore_extras: Optional[bool] = False,
extras_key: Optional[str] = None,
buffer: Optional[io.TextIOWrapper] = None,
) -> Iterable[dict]:
# Logic for handling CSV rows with more values than there are headings
for row in reader:
@ -231,6 +232,8 @@ def _extra_key_strategy(
else:
row[extras_key] = row.pop(None) # type: ignore
yield row
if buffer:
buffer.close()
def rows_from_file(
@ -299,7 +302,10 @@ def rows_from_file(
reader = csv.DictReader(decoded_fp, dialect=dialect)
else:
reader = csv.DictReader(decoded_fp)
return _extra_key_strategy(reader, ignore_extras, extras_key), Format.CSV
return (
_extra_key_strategy(reader, ignore_extras, extras_key, decoded_fp),
Format.CSV,
)
elif format == Format.TSV:
rows = rows_from_file(
fp, format=Format.CSV, dialect=csv.excel_tab, encoding=encoding