diff --git a/docs/cli.rst b/docs/cli.rst index ad752a6..7bc7a3b 100644 --- a/docs/cli.rst +++ b/docs/cli.rst @@ -975,7 +975,7 @@ Various built-in recipe functions are available for common operations. These are These recipes can be used in the code passed to ``sqlite-utils convert`` like this:: - $ sqlite-utils convert my.db mytable mycolumn \\ + $ sqlite-utils convert my.db mytable mycolumn \ 'r.jsonsplit(value, delimiter=":")' .. _cli_convert_output: diff --git a/docs/python-api.rst b/docs/python-api.rst index 7f3cdd5..7f490ab 100644 --- a/docs/python-api.rst +++ b/docs/python-api.rst @@ -702,7 +702,7 @@ You can delete all records in a table that match a specific WHERE statement usin >>> db = sqlite_utils.Database("dogs.db") >>> # Delete every dog with age less than 3 - >>> db["dogs"].delete_where("age < ?", [3]): + >>> db["dogs"].delete_where("age < ?", [3]) Calling ``table.delete_where()`` with no other arguments will delete every row in the table. @@ -736,6 +736,45 @@ An ``upsert_all()`` method is also available, which behaves like ``insert_all()` .. note:: ``.upsert()`` and ``.upsert_all()`` in sqlite-utils 1.x worked like ``.insert(..., replace=True)`` and ``.insert_all(..., replace=True)`` do in 2.x. See `issue #66 `__ for details of this change. +.. _python_api_convert: + +Converting data in columns +========================== + +The ``table.convert(...)`` method can be used to apply a conversion function to the values in a column, either to update that column or to populate new columns. It is the Python library equivalent of the :ref:`sqlite-utils convert ` command. + +This feature works by registering a custom SQLite function that applies a Python transformation, then running a SQL query equivalent to ``UPDATE table SET column = convert_value(column);`` + +To transform a specific column to uppercase, you would use the following: + +.. code-block:: python + + db["dogs"].convert("name", lambda value: value.upper()) + +You can pass a list of columns, in which case the transformation will be applied to each one: + +.. code-block:: python + + db["dogs"].convert(["name", "twitter"], lambda value: value.upper()) + +To save the output to of the transformation to a different column, use the ``output=`` parameter: + +.. code-block:: python + + db["dogs"].convert("name", lambda value: value.upper(), output="name_upper") + +This will add the new column, if it does not already exist. You can pass ``output_type=int`` or some other type to control the type of the new column - otherwise it will default to text. + +If you want to drop the original column after saving the results in a separate output column, pass ``drop=True``. + +You can create multiple new columns from a single input column by passing ``multi=True`` and a conversion function that returns a Python dictionary. This example creates new ``upper`` and ``lower`` columns populated from the single ``title`` column: + +.. code-block:: python + + table.convert( + "title", lambda v: {"upper": v.upper(), "lower": v.lower()}, multi=True + ) + .. _python_api_lookup_tables: Working with lookup tables diff --git a/sqlite_utils/db.py b/sqlite_utils/db.py index 59c6998..eb714e5 100644 --- a/sqlite_utils/db.py +++ b/sqlite_utils/db.py @@ -1729,26 +1729,26 @@ class Table(Queryable): columns[0], fn, drop=drop, show_progress=show_progress ) - todo_count = self.count * len(columns) - if output is not None: + assert len(columns) == 1, "output= can only be used with a single column" if output not in self.columns_dict: self.add_column(output, output_type or "text") + todo_count = self.count * len(columns) with progressbar(length=todo_count, silent=not show_progress) as bar: - def transform_value(v): + def convert_value(v): bar.update(1) if not v: return v return fn(v) - self.db.register_function(transform_value) + self.db.register_function(convert_value) sql = "update [{table}] set {sets};".format( table=self.name, sets=", ".join( [ - "[{output_column}] = transform_value([{column}])".format( + "[{output_column}] = convert_value([{column}])".format( output_column=output or column, column=column ) for column in columns diff --git a/tests/test_convert.py b/tests/test_convert.py index 15e8026..34e98f1 100644 --- a/tests/test_convert.py +++ b/tests/test_convert.py @@ -38,6 +38,13 @@ def test_convert_output(fresh_db, drop, expected): assert list(table.rows) == [expected] +def test_convert_output_multiple_column_error(fresh_db): + table = fresh_db["table"] + with pytest.raises(AssertionError) as excinfo: + table.convert(["title", "other"], lambda v: v, output="out") + assert "output= can only be used with a single column" in str(excinfo.value) + + @pytest.mark.parametrize( "type,expected", (