diff --git a/PKG-INFO b/PKG-INFO
index 3f39cd81..e493978d 100644
--- a/PKG-INFO
+++ b/PKG-INFO
@@ -1,213 +1,224 @@
 Metadata-Version: 2.1
 Name: swh.storage
-Version: 0.18.0
+Version: 0.19.0
 Summary: Software Heritage storage manager
 Home-page: https://forge.softwareheritage.org/diffusion/DSTO/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 License: UNKNOWN
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-storage
 Project-URL: Documentation, https://docs.softwareheritage.org/devel/swh-storage/
 Description: swh-storage
         ===========
         
         Abstraction layer over the archive, allowing to access all stored source code
         artifacts as well as their metadata.
         
         See the
         [documentation](https://docs.softwareheritage.org/devel/swh-storage/index.html)
         for more details.
         
         ## Quick start
         
         ### Dependencies
         
         Python tests for this module include tests that cannot be run without a local
         Postgresql database, so you need the Postgresql server executable on your
         machine (no need to have a running Postgresql server). They also expect a
         cassandra server.
         
         #### Debian-like host
         
         ```
         $ sudo apt install libpq-dev postgresql-11 cassandra
         ```
         
         #### Non Debian-like host
         
         The tests expects the path to `cassandra` to either be unspecified, it is then
         looked up at `/usr/sbin/cassandra`, either specified through the environment
         variable `SWH_CASSANDRA_BIN`.
         
         Optionally, you can avoid running the cassandra tests.
         
         ```
         (swh) :~/swh-storage$ tox -- -m 'not cassandra'
         ```
         
         ### Installation
         
         It is strongly recommended to use a virtualenv. In the following, we
         consider you work in a virtualenv named `swh`. See the
         [developer setup guide](https://docs.softwareheritage.org/devel/developer-setup.html#developer-setup)
         for a more details on how to setup a working environment.
         
         
         You can install the package directly from
         [pypi](https://pypi.org/p/swh.storage):
         
         ```
         (swh) :~$ pip install swh.storage
         [...]
         ```
         
         Or from sources:
         
         ```
         (swh) :~$ git clone https://forge.softwareheritage.org/source/swh-storage.git
         [...]
         (swh) :~$ cd swh-storage
         (swh) :~/swh-storage$ pip install .
         [...]
         ```
         
         Then you can check it's properly installed:
         ```
         (swh) :~$ swh storage --help
         Usage: swh storage [OPTIONS] COMMAND [ARGS]...
         
           Software Heritage Storage tools.
         
         Options:
           -h, --help  Show this message and exit.
         
         Commands:
           rpc-serve  Software Heritage Storage RPC server.
         ```
         
         
         ## Tests
         
         The best way of running Python tests for this module is to use
         [tox](https://tox.readthedocs.io/).
         
         ```
         (swh) :~$ pip install tox
         ```
         
         ### tox
         
         From the sources directory, simply use tox:
         
         ```
         (swh) :~/swh-storage$ tox
         [...]
         ========= 315 passed, 6 skipped, 15 warnings in 40.86 seconds ==========
         _______________________________ summary ________________________________
           flake8: commands succeeded
           py3: commands succeeded
           congratulations :)
         ```
         
+        Note: it is possible to set the `JAVA_HOME` environment variable to specify the
+        version of the JVM to be used by Cassandra. For example, at the time of writing
+        this, Cassandra does not support java 14, so one may want to use for example
+        java 11:
+        
+        ```
+        (swh) :~/swh-storage$ export JAVA_HOME=/usr/lib/jvm/java-14-openjdk-amd64/bin/java
+        (swh) :~/swh-storage$ tox
+        [...]
+        ```
+        
         ## Development
         
         The storage server can be locally started. It requires a configuration file and
         a running Postgresql database.
         
         ### Sample configuration
         
         A typical configuration `storage.yml` file is:
         
         ```
         storage:
           cls: local
           db: "dbname=softwareheritage-dev user=<user> password=<pwd>"
           objstorage:
             cls: pathslicing
             root: /tmp/swh-storage/
             slicing: 0:2/2:4/4:6
         ```
         
         which means, this uses:
         
         - a local storage instance whose db connection is to
           `softwareheritage-dev` local instance,
         
         - the objstorage uses a local objstorage instance whose:
         
           - `root` path is /tmp/swh-storage,
         
           - slicing scheme is `0:2/2:4/4:6`. This means that the identifier of
             the content (sha1) which will be stored on disk at first level
             with the first 2 hex characters, the second level with the next 2
             hex characters and the third level with the next 2 hex
             characters. And finally the complete hash file holding the raw
             content. For example: 00062f8bd330715c4f819373653d97b3cd34394c
             will be stored at 00/06/2f/00062f8bd330715c4f819373653d97b3cd34394c
         
         Note that the `root` path should exist on disk before starting the server.
         
         
         ### Starting the storage server
         
         If the python package has been properly installed (e.g. in a virtual env), you
         should be able to use the command:
         
         ```
         (swh) :~/swh-storage$ swh storage rpc-serve storage.yml
         ```
         
         This runs a local swh-storage api at 5002 port.
         
         ```
         (swh) :~/swh-storage$ curl http://127.0.0.1:5002
         <html>
         <head><title>Software Heritage storage server</title></head>
         <body>
         <p>You have reached the
         <a href="https://www.softwareheritage.org/">Software Heritage</a>
         storage server.<br />
         See its
         <a href="https://docs.softwareheritage.org/devel/swh-storage/">documentation
         and API</a> for more information</p>
         ```
         
         ### And then what?
         
         In your upper layer
         ([loader-git](https://forge.softwareheritage.org/source/swh-loader-git/),
         [loader-svn](https://forge.softwareheritage.org/source/swh-loader-svn/),
         etc...), you can define a remote storage with this snippet of yaml
         configuration.
         
         ```
         storage:
           cls: remote
           url: http://localhost:5002/
         ```
         
         You could directly define a local storage with the following snippet:
         
         ```
         storage:
           cls: local
           db: service=swh-dev
           objstorage:
             cls: pathslicing
             root: /home/storage/swh-storage/
             slicing: 0:2/2:4/4:6
         ```
         
 Platform: UNKNOWN
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Requires-Python: >=3.7
 Description-Content-Type: text/markdown
 Provides-Extra: testing
 Provides-Extra: schemata
 Provides-Extra: journal
diff --git a/README.md b/README.md
index 5bcc7a42..4aee17fa 100644
--- a/README.md
+++ b/README.md
@@ -1,189 +1,200 @@
 swh-storage
 ===========
 
 Abstraction layer over the archive, allowing to access all stored source code
 artifacts as well as their metadata.
 
 See the
 [documentation](https://docs.softwareheritage.org/devel/swh-storage/index.html)
 for more details.
 
 ## Quick start
 
 ### Dependencies
 
 Python tests for this module include tests that cannot be run without a local
 Postgresql database, so you need the Postgresql server executable on your
 machine (no need to have a running Postgresql server). They also expect a
 cassandra server.
 
 #### Debian-like host
 
 ```
 $ sudo apt install libpq-dev postgresql-11 cassandra
 ```
 
 #### Non Debian-like host
 
 The tests expects the path to `cassandra` to either be unspecified, it is then
 looked up at `/usr/sbin/cassandra`, either specified through the environment
 variable `SWH_CASSANDRA_BIN`.
 
 Optionally, you can avoid running the cassandra tests.
 
 ```
 (swh) :~/swh-storage$ tox -- -m 'not cassandra'
 ```
 
 ### Installation
 
 It is strongly recommended to use a virtualenv. In the following, we
 consider you work in a virtualenv named `swh`. See the
 [developer setup guide](https://docs.softwareheritage.org/devel/developer-setup.html#developer-setup)
 for a more details on how to setup a working environment.
 
 
 You can install the package directly from
 [pypi](https://pypi.org/p/swh.storage):
 
 ```
 (swh) :~$ pip install swh.storage
 [...]
 ```
 
 Or from sources:
 
 ```
 (swh) :~$ git clone https://forge.softwareheritage.org/source/swh-storage.git
 [...]
 (swh) :~$ cd swh-storage
 (swh) :~/swh-storage$ pip install .
 [...]
 ```
 
 Then you can check it's properly installed:
 ```
 (swh) :~$ swh storage --help
 Usage: swh storage [OPTIONS] COMMAND [ARGS]...
 
   Software Heritage Storage tools.
 
 Options:
   -h, --help  Show this message and exit.
 
 Commands:
   rpc-serve  Software Heritage Storage RPC server.
 ```
 
 
 ## Tests
 
 The best way of running Python tests for this module is to use
 [tox](https://tox.readthedocs.io/).
 
 ```
 (swh) :~$ pip install tox
 ```
 
 ### tox
 
 From the sources directory, simply use tox:
 
 ```
 (swh) :~/swh-storage$ tox
 [...]
 ========= 315 passed, 6 skipped, 15 warnings in 40.86 seconds ==========
 _______________________________ summary ________________________________
   flake8: commands succeeded
   py3: commands succeeded
   congratulations :)
 ```
 
+Note: it is possible to set the `JAVA_HOME` environment variable to specify the
+version of the JVM to be used by Cassandra. For example, at the time of writing
+this, Cassandra does not support java 14, so one may want to use for example
+java 11:
+
+```
+(swh) :~/swh-storage$ export JAVA_HOME=/usr/lib/jvm/java-14-openjdk-amd64/bin/java
+(swh) :~/swh-storage$ tox
+[...]
+```
+
 ## Development
 
 The storage server can be locally started. It requires a configuration file and
 a running Postgresql database.
 
 ### Sample configuration
 
 A typical configuration `storage.yml` file is:
 
 ```
 storage:
   cls: local
   db: "dbname=softwareheritage-dev user=<user> password=<pwd>"
   objstorage:
     cls: pathslicing
     root: /tmp/swh-storage/
     slicing: 0:2/2:4/4:6
 ```
 
 which means, this uses:
 
 - a local storage instance whose db connection is to
   `softwareheritage-dev` local instance,
 
 - the objstorage uses a local objstorage instance whose:
 
   - `root` path is /tmp/swh-storage,
 
   - slicing scheme is `0:2/2:4/4:6`. This means that the identifier of
     the content (sha1) which will be stored on disk at first level
     with the first 2 hex characters, the second level with the next 2
     hex characters and the third level with the next 2 hex
     characters. And finally the complete hash file holding the raw
     content. For example: 00062f8bd330715c4f819373653d97b3cd34394c
     will be stored at 00/06/2f/00062f8bd330715c4f819373653d97b3cd34394c
 
 Note that the `root` path should exist on disk before starting the server.
 
 
 ### Starting the storage server
 
 If the python package has been properly installed (e.g. in a virtual env), you
 should be able to use the command:
 
 ```
 (swh) :~/swh-storage$ swh storage rpc-serve storage.yml
 ```
 
 This runs a local swh-storage api at 5002 port.
 
 ```
 (swh) :~/swh-storage$ curl http://127.0.0.1:5002
 <html>
 <head><title>Software Heritage storage server</title></head>
 <body>
 <p>You have reached the
 <a href="https://www.softwareheritage.org/">Software Heritage</a>
 storage server.<br />
 See its
 <a href="https://docs.softwareheritage.org/devel/swh-storage/">documentation
 and API</a> for more information</p>
 ```
 
 ### And then what?
 
 In your upper layer
 ([loader-git](https://forge.softwareheritage.org/source/swh-loader-git/),
 [loader-svn](https://forge.softwareheritage.org/source/swh-loader-svn/),
 etc...), you can define a remote storage with this snippet of yaml
 configuration.
 
 ```
 storage:
   cls: remote
   url: http://localhost:5002/
 ```
 
 You could directly define a local storage with the following snippet:
 
 ```
 storage:
   cls: local
   db: service=swh-dev
   objstorage:
     cls: pathslicing
     root: /home/storage/swh-storage/
     slicing: 0:2/2:4/4:6
 ```
diff --git a/conftest.py b/conftest.py
index 474099e2..8b9322e5 100644
--- a/conftest.py
+++ b/conftest.py
@@ -1,6 +1,6 @@
 # Copyright (C) 2020  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
-pytest_plugins = ["swh.storage.pytest_plugin"]
+pytest_plugins = ["swh.storage.pytest_plugin", "swh.core.db.pytest_plugin"]
diff --git a/docs/cli.rst b/docs/cli.rst
new file mode 100644
index 00000000..74e78cc2
--- /dev/null
+++ b/docs/cli.rst
@@ -0,0 +1,8 @@
+.. _swh-storage-cli:
+
+Command-line interface
+======================
+
+.. click:: swh.storage.cli:storage
+  :prog: swh storage
+  :nested: full
diff --git a/docs/index.rst b/docs/index.rst
index 502967a3..17037bbd 100644
--- a/docs/index.rst
+++ b/docs/index.rst
@@ -1,45 +1,46 @@
 .. _swh-storage:
 
 Software Heritage - Storage
 ===========================
 
 Abstraction layer over the archive, allowing to access all stored source code
 artifacts as well as their metadata
 
 
 The Software Heritage storage consist of a high-level storage layer
 (:mod:`swh.storage`) that exposes a client/server API
 (:mod:`swh.storage.api`). The API is exposed by a server
 (:mod:`swh.storage.api.server`) and accessible via a client
 (:mod:`swh.storage.api.client`).
 
 The low-level implementation of the storage is split between an object storage
 (:ref:`swh.objstorage <swh-objstorage>`), which stores all "blobs" (i.e., the
 leaves of the :ref:`data-model`) and a SQL representation of the rest of the
 graph (:mod:`swh.storage.storage`).
 
 
 Database schema
 ---------------
 
 * :ref:`sql-storage`
 
 
 Archive copies
 --------------
 
 * :ref:`archive-copies`
 
 Specifications
 --------------
 
 * :ref:`extrinsic-metadata-specification`
 
 
 Reference Documentation
 -----------------------
 
 .. toctree::
    :maxdepth: 2
 
+   cli
    /apidoc/swh.storage
diff --git a/requirements-test.txt b/requirements-test.txt
index 07dc055f..310fc35e 100644
--- a/requirements-test.txt
+++ b/requirements-test.txt
@@ -1,10 +1,10 @@
-hypothesis >= 3.11.0
+hypothesis >= 3.11.0, < 6
 pytest
 pytest-mock
 sqlalchemy-stubs
 # pytz is in fact a dep of swh.model[testing] and should not be necessary, but
 # the dep on swh.model in the main requirements-swh.txt file shadows this one
 # adding the [testing] extra.
 swh.model[testing] >= 0.0.50
 pytz
 pytest-xdist
diff --git a/swh.storage.egg-info/PKG-INFO b/swh.storage.egg-info/PKG-INFO
index 3f39cd81..e493978d 100644
--- a/swh.storage.egg-info/PKG-INFO
+++ b/swh.storage.egg-info/PKG-INFO
@@ -1,213 +1,224 @@
 Metadata-Version: 2.1
 Name: swh.storage
-Version: 0.18.0
+Version: 0.19.0
 Summary: Software Heritage storage manager
 Home-page: https://forge.softwareheritage.org/diffusion/DSTO/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 License: UNKNOWN
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-storage
 Project-URL: Documentation, https://docs.softwareheritage.org/devel/swh-storage/
 Description: swh-storage
         ===========
         
         Abstraction layer over the archive, allowing to access all stored source code
         artifacts as well as their metadata.
         
         See the
         [documentation](https://docs.softwareheritage.org/devel/swh-storage/index.html)
         for more details.
         
         ## Quick start
         
         ### Dependencies
         
         Python tests for this module include tests that cannot be run without a local
         Postgresql database, so you need the Postgresql server executable on your
         machine (no need to have a running Postgresql server). They also expect a
         cassandra server.
         
         #### Debian-like host
         
         ```
         $ sudo apt install libpq-dev postgresql-11 cassandra
         ```
         
         #### Non Debian-like host
         
         The tests expects the path to `cassandra` to either be unspecified, it is then
         looked up at `/usr/sbin/cassandra`, either specified through the environment
         variable `SWH_CASSANDRA_BIN`.
         
         Optionally, you can avoid running the cassandra tests.
         
         ```
         (swh) :~/swh-storage$ tox -- -m 'not cassandra'
         ```
         
         ### Installation
         
         It is strongly recommended to use a virtualenv. In the following, we
         consider you work in a virtualenv named `swh`. See the
         [developer setup guide](https://docs.softwareheritage.org/devel/developer-setup.html#developer-setup)
         for a more details on how to setup a working environment.
         
         
         You can install the package directly from
         [pypi](https://pypi.org/p/swh.storage):
         
         ```
         (swh) :~$ pip install swh.storage
         [...]
         ```
         
         Or from sources:
         
         ```
         (swh) :~$ git clone https://forge.softwareheritage.org/source/swh-storage.git
         [...]
         (swh) :~$ cd swh-storage
         (swh) :~/swh-storage$ pip install .
         [...]
         ```
         
         Then you can check it's properly installed:
         ```
         (swh) :~$ swh storage --help
         Usage: swh storage [OPTIONS] COMMAND [ARGS]...
         
           Software Heritage Storage tools.
         
         Options:
           -h, --help  Show this message and exit.
         
         Commands:
           rpc-serve  Software Heritage Storage RPC server.
         ```
         
         
         ## Tests
         
         The best way of running Python tests for this module is to use
         [tox](https://tox.readthedocs.io/).
         
         ```
         (swh) :~$ pip install tox
         ```
         
         ### tox
         
         From the sources directory, simply use tox:
         
         ```
         (swh) :~/swh-storage$ tox
         [...]
         ========= 315 passed, 6 skipped, 15 warnings in 40.86 seconds ==========
         _______________________________ summary ________________________________
           flake8: commands succeeded
           py3: commands succeeded
           congratulations :)
         ```
         
+        Note: it is possible to set the `JAVA_HOME` environment variable to specify the
+        version of the JVM to be used by Cassandra. For example, at the time of writing
+        this, Cassandra does not support java 14, so one may want to use for example
+        java 11:
+        
+        ```
+        (swh) :~/swh-storage$ export JAVA_HOME=/usr/lib/jvm/java-14-openjdk-amd64/bin/java
+        (swh) :~/swh-storage$ tox
+        [...]
+        ```
+        
         ## Development
         
         The storage server can be locally started. It requires a configuration file and
         a running Postgresql database.
         
         ### Sample configuration
         
         A typical configuration `storage.yml` file is:
         
         ```
         storage:
           cls: local
           db: "dbname=softwareheritage-dev user=<user> password=<pwd>"
           objstorage:
             cls: pathslicing
             root: /tmp/swh-storage/
             slicing: 0:2/2:4/4:6
         ```
         
         which means, this uses:
         
         - a local storage instance whose db connection is to
           `softwareheritage-dev` local instance,
         
         - the objstorage uses a local objstorage instance whose:
         
           - `root` path is /tmp/swh-storage,
         
           - slicing scheme is `0:2/2:4/4:6`. This means that the identifier of
             the content (sha1) which will be stored on disk at first level
             with the first 2 hex characters, the second level with the next 2
             hex characters and the third level with the next 2 hex
             characters. And finally the complete hash file holding the raw
             content. For example: 00062f8bd330715c4f819373653d97b3cd34394c
             will be stored at 00/06/2f/00062f8bd330715c4f819373653d97b3cd34394c
         
         Note that the `root` path should exist on disk before starting the server.
         
         
         ### Starting the storage server
         
         If the python package has been properly installed (e.g. in a virtual env), you
         should be able to use the command:
         
         ```
         (swh) :~/swh-storage$ swh storage rpc-serve storage.yml
         ```
         
         This runs a local swh-storage api at 5002 port.
         
         ```
         (swh) :~/swh-storage$ curl http://127.0.0.1:5002
         <html>
         <head><title>Software Heritage storage server</title></head>
         <body>
         <p>You have reached the
         <a href="https://www.softwareheritage.org/">Software Heritage</a>
         storage server.<br />
         See its
         <a href="https://docs.softwareheritage.org/devel/swh-storage/">documentation
         and API</a> for more information</p>
         ```
         
         ### And then what?
         
         In your upper layer
         ([loader-git](https://forge.softwareheritage.org/source/swh-loader-git/),
         [loader-svn](https://forge.softwareheritage.org/source/swh-loader-svn/),
         etc...), you can define a remote storage with this snippet of yaml
         configuration.
         
         ```
         storage:
           cls: remote
           url: http://localhost:5002/
         ```
         
         You could directly define a local storage with the following snippet:
         
         ```
         storage:
           cls: local
           db: service=swh-dev
           objstorage:
             cls: pathslicing
             root: /home/storage/swh-storage/
             slicing: 0:2/2:4/4:6
         ```
         
 Platform: UNKNOWN
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Requires-Python: >=3.7
 Description-Content-Type: text/markdown
 Provides-Extra: testing
 Provides-Extra: schemata
 Provides-Extra: journal
diff --git a/swh.storage.egg-info/SOURCES.txt b/swh.storage.egg-info/SOURCES.txt
index 9c7291fd..205d9c33 100644
--- a/swh.storage.egg-info/SOURCES.txt
+++ b/swh.storage.egg-info/SOURCES.txt
@@ -1,313 +1,314 @@
 .gitignore
 .pre-commit-config.yaml
 AUTHORS
 CODE_OF_CONDUCT.md
 CONTRIBUTORS
 LICENSE
 MANIFEST.in
 Makefile
 Makefile.local
 README.md
 conftest.py
 mypy.ini
 pyproject.toml
 pytest.ini
 requirements-swh-journal.txt
 requirements-swh.txt
 requirements-test.txt
 requirements.txt
 setup.cfg
 setup.py
 tox.ini
 ./requirements-swh-journal.txt
 ./requirements-swh.txt
 ./requirements-test.txt
 ./requirements.txt
 bin/swh-storage-add-dir
 docs/.gitignore
 docs/Makefile
 docs/Makefile.local
 docs/archive-copies.rst
+docs/cli.rst
 docs/conf.py
 docs/extrinsic-metadata-specification.rst
 docs/index.rst
 docs/sql-storage.rst
 docs/_static/.placeholder
 docs/_templates/.placeholder
 docs/images/.gitignore
 docs/images/Makefile
 docs/images/swh-archive-copies.dia
 sql/.gitignore
 sql/Makefile
 sql/TODO
 sql/clusters.dot
 sql/bin/db-upgrade
 sql/bin/dot_add_content
 sql/doc/json
 sql/doc/json/.gitignore
 sql/doc/json/Makefile
 sql/doc/json/entity.lister_metadata.schema.json
 sql/doc/json/entity.metadata.schema.json
 sql/doc/json/entity_history.lister_metadata.schema.json
 sql/doc/json/entity_history.metadata.schema.json
 sql/doc/json/fetch_history.result.schema.json
 sql/doc/json/list_history.result.schema.json
 sql/doc/json/listable_entity.list_params.schema.json
 sql/doc/json/origin_visit.metadata.json
 sql/doc/json/tool.tool_configuration.schema.json
 sql/json/.gitignore
 sql/json/Makefile
 sql/json/entity.lister_metadata.schema.json
 sql/json/entity.metadata.schema.json
 sql/json/entity_history.lister_metadata.schema.json
 sql/json/entity_history.metadata.schema.json
 sql/json/fetch_history.result.schema.json
 sql/json/list_history.result.schema.json
 sql/json/listable_entity.list_params.schema.json
 sql/json/origin_visit.metadata.json
 sql/json/tool.tool_configuration.schema.json
 sql/upgrades/015.sql
 sql/upgrades/016.sql
 sql/upgrades/017.sql
 sql/upgrades/018.sql
 sql/upgrades/019.sql
 sql/upgrades/020.sql
 sql/upgrades/021.sql
 sql/upgrades/022.sql
 sql/upgrades/023.sql
 sql/upgrades/024.sql
 sql/upgrades/025.sql
 sql/upgrades/026.sql
 sql/upgrades/027.sql
 sql/upgrades/028.sql
 sql/upgrades/029.sql
 sql/upgrades/030.sql
 sql/upgrades/032.sql
 sql/upgrades/033.sql
 sql/upgrades/034.sql
 sql/upgrades/035.sql
 sql/upgrades/036.sql
 sql/upgrades/037.sql
 sql/upgrades/038.sql
 sql/upgrades/039.sql
 sql/upgrades/040.sql
 sql/upgrades/041.sql
 sql/upgrades/042.sql
 sql/upgrades/043.sql
 sql/upgrades/044.sql
 sql/upgrades/045.sql
 sql/upgrades/046.sql
 sql/upgrades/047.sql
 sql/upgrades/048.sql
 sql/upgrades/049.sql
 sql/upgrades/050.sql
 sql/upgrades/051.sql
 sql/upgrades/052.sql
 sql/upgrades/053.sql
 sql/upgrades/054.sql
 sql/upgrades/055.sql
 sql/upgrades/056.sql
 sql/upgrades/057.sql
 sql/upgrades/058.sql
 sql/upgrades/059.sql
 sql/upgrades/060.sql
 sql/upgrades/061.sql
 sql/upgrades/062.sql
 sql/upgrades/063.sql
 sql/upgrades/064.sql
 sql/upgrades/065.sql
 sql/upgrades/066.sql
 sql/upgrades/067.sql
 sql/upgrades/068.sql
 sql/upgrades/069.sql
 sql/upgrades/070.sql
 sql/upgrades/071.sql
 sql/upgrades/072.sql
 sql/upgrades/073.sql
 sql/upgrades/074.sql
 sql/upgrades/075.sql
 sql/upgrades/076.sql
 sql/upgrades/077.sql
 sql/upgrades/078.sql
 sql/upgrades/079.sql
 sql/upgrades/080.sql
 sql/upgrades/081.sql
 sql/upgrades/082.sql
 sql/upgrades/083.sql
 sql/upgrades/084.sql
 sql/upgrades/085.sql
 sql/upgrades/086.sql
 sql/upgrades/087.sql
 sql/upgrades/088.sql
 sql/upgrades/089.sql
 sql/upgrades/090.sql
 sql/upgrades/091.sql
 sql/upgrades/092.sql
 sql/upgrades/093.sql
 sql/upgrades/094.sql
 sql/upgrades/095.sql
 sql/upgrades/096.sql
 sql/upgrades/097.sql
 sql/upgrades/098.sql
 sql/upgrades/099.sql
 sql/upgrades/100.sql
 sql/upgrades/101.sql
 sql/upgrades/102.sql
 sql/upgrades/103.sql
 sql/upgrades/104.sql
 sql/upgrades/105.sql
 sql/upgrades/106.sql
 sql/upgrades/107.sql
 sql/upgrades/108.sql
 sql/upgrades/109.sql
 sql/upgrades/110.sql
 sql/upgrades/111.sql
 sql/upgrades/112.sql
 sql/upgrades/113.sql
 sql/upgrades/114.sql
 sql/upgrades/115.sql
 sql/upgrades/116.sql
 sql/upgrades/117.sql
 sql/upgrades/118.sql
 sql/upgrades/119.sql
 sql/upgrades/120.sql
 sql/upgrades/121.sql
 sql/upgrades/122.sql
 sql/upgrades/123.sql
 sql/upgrades/124.sql
 sql/upgrades/125.sql
 sql/upgrades/126.sql
 sql/upgrades/127.sql
 sql/upgrades/128.sql
 sql/upgrades/129.sql
 sql/upgrades/130.sql
 sql/upgrades/131.sql
 sql/upgrades/132.sql
 sql/upgrades/133.sql
 sql/upgrades/134.sql
 sql/upgrades/135.sql
 sql/upgrades/136.sql
 sql/upgrades/137.sql
 sql/upgrades/138.sql
 sql/upgrades/139.sql
 sql/upgrades/140.sql
 sql/upgrades/141.sql
 sql/upgrades/142.sql
 sql/upgrades/143.sql
 sql/upgrades/144.sql
 sql/upgrades/145.sql
 sql/upgrades/146.sql
 sql/upgrades/147.sql
 sql/upgrades/148.sql
 sql/upgrades/149.sql
 sql/upgrades/150.sql
 sql/upgrades/151.sql
 sql/upgrades/152.sql
 sql/upgrades/153.sql
 sql/upgrades/154.sql
 sql/upgrades/155.sql
 sql/upgrades/156.sql
 sql/upgrades/157.sql
 sql/upgrades/158.sql
 sql/upgrades/159.sql
 sql/upgrades/160.sql
 sql/upgrades/161.sql
 sql/upgrades/162.sql
 sql/upgrades/163.sql
 sql/upgrades/164.sql
 swh/__init__.py
 swh.storage.egg-info/PKG-INFO
 swh.storage.egg-info/SOURCES.txt
 swh.storage.egg-info/dependency_links.txt
 swh.storage.egg-info/entry_points.txt
 swh.storage.egg-info/requires.txt
 swh.storage.egg-info/top_level.txt
 swh/storage/__init__.py
 swh/storage/backfill.py
 swh/storage/buffer.py
 swh/storage/cli.py
 swh/storage/common.py
 swh/storage/exc.py
 swh/storage/filter.py
 swh/storage/fixer.py
 swh/storage/in_memory.py
 swh/storage/interface.py
 swh/storage/metrics.py
 swh/storage/migrate_extrinsic_metadata.py
 swh/storage/objstorage.py
 swh/storage/py.typed
 swh/storage/pytest_plugin.py
 swh/storage/replay.py
 swh/storage/retry.py
 swh/storage/utils.py
 swh/storage/validate.py
 swh/storage/writer.py
 swh/storage/algos/__init__.py
 swh/storage/algos/diff.py
 swh/storage/algos/dir_iterators.py
 swh/storage/algos/origin.py
 swh/storage/algos/revisions_walker.py
 swh/storage/algos/snapshot.py
 swh/storage/api/__init__.py
 swh/storage/api/client.py
 swh/storage/api/serializers.py
 swh/storage/api/server.py
 swh/storage/cassandra/__init__.py
 swh/storage/cassandra/common.py
 swh/storage/cassandra/converters.py
 swh/storage/cassandra/cql.py
 swh/storage/cassandra/model.py
 swh/storage/cassandra/schema.py
 swh/storage/cassandra/storage.py
 swh/storage/postgresql/__init__.py
 swh/storage/postgresql/converters.py
 swh/storage/postgresql/db.py
 swh/storage/postgresql/storage.py
 swh/storage/sql/10-superuser-init.sql
 swh/storage/sql/15-flavor.sql
 swh/storage/sql/20-enums.sql
 swh/storage/sql/30-schema.sql
 swh/storage/sql/40-funcs.sql
 swh/storage/sql/60-indexes.sql
 swh/storage/sql/logical_replication/replication_source.sql
 swh/storage/tests/__init__.py
 swh/storage/tests/conftest.py
 swh/storage/tests/storage_data.py
 swh/storage/tests/storage_tests.py
 swh/storage/tests/test_api_client.py
 swh/storage/tests/test_backfill.py
 swh/storage/tests/test_buffer.py
 swh/storage/tests/test_cassandra.py
 swh/storage/tests/test_cassandra_converters.py
 swh/storage/tests/test_cli.py
 swh/storage/tests/test_exception.py
 swh/storage/tests/test_filter.py
 swh/storage/tests/test_in_memory.py
 swh/storage/tests/test_init.py
 swh/storage/tests/test_kafka_writer.py
 swh/storage/tests/test_metrics.py
 swh/storage/tests/test_postgresql.py
 swh/storage/tests/test_postgresql_converters.py
 swh/storage/tests/test_pytest_plugin.py
 swh/storage/tests/test_replay.py
 swh/storage/tests/test_retry.py
 swh/storage/tests/test_revision_bw_compat.py
 swh/storage/tests/test_serializers.py
 swh/storage/tests/test_server.py
 swh/storage/tests/test_storage_data.py
 swh/storage/tests/test_utils.py
 swh/storage/tests/test_validate.py
 swh/storage/tests/algos/__init__.py
 swh/storage/tests/algos/test_diff.py
 swh/storage/tests/algos/test_dir_iterator.py
 swh/storage/tests/algos/test_origin.py
 swh/storage/tests/algos/test_revisions_walker.py
 swh/storage/tests/algos/test_snapshot.py
 swh/storage/tests/data/storage.yml
 swh/storage/tests/migrate_extrinsic_metadata/test_cran.py
 swh/storage/tests/migrate_extrinsic_metadata/test_debian.py
 swh/storage/tests/migrate_extrinsic_metadata/test_deposit.py
 swh/storage/tests/migrate_extrinsic_metadata/test_gnu.py
 swh/storage/tests/migrate_extrinsic_metadata/test_nixguix.py
 swh/storage/tests/migrate_extrinsic_metadata/test_npm.py
 swh/storage/tests/migrate_extrinsic_metadata/test_pypi.py
\ No newline at end of file
diff --git a/swh.storage.egg-info/requires.txt b/swh.storage.egg-info/requires.txt
index 2999362d..233a9a60 100644
--- a/swh.storage.egg-info/requires.txt
+++ b/swh.storage.egg-info/requires.txt
@@ -1,29 +1,29 @@
 click
 flask
 psycopg2
 aiohttp
 tenacity
 cassandra-driver!=3.21.0,>=3.19.0
 deprecated
 typing-extensions
 mypy_extensions
 iso8601
 swh.core[db,http]>=0.5
 swh.model>=0.7.2
 swh.objstorage>=0.2.2
 
 [journal]
 swh.journal>=0.5.1
 
 [schemata]
 SQLAlchemy
 
 [testing]
-hypothesis>=3.11.0
+hypothesis<6,>=3.11.0
 pytest
 pytest-mock
 sqlalchemy-stubs
 swh.model[testing]>=0.0.50
 pytz
 pytest-xdist
 swh.journal>=0.5.1
diff --git a/swh/storage/backfill.py b/swh/storage/backfill.py
index 065a8ca0..17705136 100644
--- a/swh/storage/backfill.py
+++ b/swh/storage/backfill.py
@@ -1,554 +1,554 @@
 # Copyright (C) 2017-2020  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 """Storage backfiller.
 
 The backfiller goal is to produce back part or all of the objects
 from a storage to the journal topics
 
 Current implementation consists in the JournalBackfiller class.
 
 It simply reads the objects from the storage and sends every object identifier back to
 the journal.
 
 """
 
 import logging
 from typing import Any, Callable, Dict, Optional
 
 from swh.core.db import BaseDb
 from swh.model.model import (
     BaseModel,
     Directory,
     DirectoryEntry,
     RawExtrinsicMetadata,
     Release,
     Revision,
     Snapshot,
     SnapshotBranch,
     TargetType,
 )
 from swh.storage.postgresql.converters import (
     db_to_raw_extrinsic_metadata,
     db_to_release,
     db_to_revision,
 )
 from swh.storage.replay import object_converter_fn
 from swh.storage.writer import JournalWriter
 
 logger = logging.getLogger(__name__)
 
 PARTITION_KEY = {
     "content": "sha1",
     "skipped_content": "sha1",
     "directory": "id",
     "metadata_authority": "type, url",
     "metadata_fetcher": "name, version",
     "raw_extrinsic_metadata": "target",
     "revision": "revision.id",
     "release": "release.id",
     "snapshot": "id",
     "origin": "id",
     "origin_visit": "origin_visit.origin",
     "origin_visit_status": "origin_visit_status.origin",
 }
 
 COLUMNS = {
     "content": [
         "sha1",
         "sha1_git",
         "sha256",
         "blake2s256",
         "length",
         "status",
         "ctime",
     ],
     "skipped_content": [
         "sha1",
         "sha1_git",
         "sha256",
         "blake2s256",
         "length",
         "ctime",
         "status",
         "reason",
     ],
     "directory": ["id", "dir_entries", "file_entries", "rev_entries"],
     "metadata_authority": ["type", "url", "metadata",],
     "metadata_fetcher": ["name", "version", "metadata",],
     "raw_extrinsic_metadata": [
         "raw_extrinsic_metadata.type",
         "raw_extrinsic_metadata.target",
         "metadata_authority.type",
         "metadata_authority.url",
         "metadata_fetcher.name",
         "metadata_fetcher.version",
         "discovery_date",
         "format",
         "raw_extrinsic_metadata.metadata",
         "origin",
         "visit",
         "snapshot",
         "release",
         "revision",
         "path",
         "directory",
     ],
     "revision": [
         ("revision.id", "id"),
         "date",
         "date_offset",
         "date_neg_utc_offset",
         "committer_date",
         "committer_date_offset",
         "committer_date_neg_utc_offset",
         "type",
         "directory",
         "message",
         "synthetic",
         "metadata",
         "extra_headers",
         (
             "array(select parent_id::bytea from revision_history rh "
             "where rh.id = revision.id order by rh.parent_rank asc)",
             "parents",
         ),
         ("a.id", "author_id"),
         ("a.name", "author_name"),
         ("a.email", "author_email"),
         ("a.fullname", "author_fullname"),
         ("c.id", "committer_id"),
         ("c.name", "committer_name"),
         ("c.email", "committer_email"),
         ("c.fullname", "committer_fullname"),
     ],
     "release": [
         ("release.id", "id"),
         "date",
         "date_offset",
         "date_neg_utc_offset",
         "comment",
         ("release.name", "name"),
         "synthetic",
         "target",
         "target_type",
         ("a.id", "author_id"),
         ("a.name", "author_name"),
         ("a.email", "author_email"),
         ("a.fullname", "author_fullname"),
     ],
     "snapshot": ["id", "object_id"],
     "origin": ["url"],
     "origin_visit": ["visit", "type", ("origin.url", "origin"), "date",],
     "origin_visit_status": [
         "visit",
         ("origin.url", "origin"),
         "date",
         "snapshot",
         "status",
         "metadata",
     ],
 }
 
 
 JOINS = {
     "release": ["person a on release.author=a.id"],
     "revision": [
         "person a on revision.author=a.id",
         "person c on revision.committer=c.id",
     ],
     "origin_visit": ["origin on origin_visit.origin=origin.id"],
     "origin_visit_status": ["origin on origin_visit_status.origin=origin.id"],
     "raw_extrinsic_metadata": [
         "metadata_authority on "
         "raw_extrinsic_metadata.authority_id=metadata_authority.id",
         "metadata_fetcher on raw_extrinsic_metadata.fetcher_id=metadata_fetcher.id",
     ],
 }
 
 
 def directory_converter(db: BaseDb, directory_d: Dict[str, Any]) -> Directory:
     """Convert directory from the flat representation to swh model
        compatible objects.
 
     """
     columns = ["target", "name", "perms"]
     query_template = """
     select %(columns)s
     from directory_entry_%(type)s
     where id in %%s
     """
 
     types = ["file", "dir", "rev"]
 
     entries = []
     with db.cursor() as cur:
         for type in types:
             ids = directory_d.pop("%s_entries" % type)
             if not ids:
                 continue
             query = query_template % {
                 "columns": ",".join(columns),
                 "type": type,
             }
             cur.execute(query, (tuple(ids),))
             for row in cur:
                 entry_d = dict(zip(columns, row))
                 entry = DirectoryEntry(
                     name=entry_d["name"],
                     type=type,
                     target=entry_d["target"],
                     perms=entry_d["perms"],
                 )
                 entries.append(entry)
 
     return Directory(id=directory_d["id"], entries=tuple(entries),)
 
 
 def raw_extrinsic_metadata_converter(
     db: BaseDb, metadata: Dict[str, Any]
 ) -> RawExtrinsicMetadata:
     """Convert revision from the flat representation to swh model
        compatible objects.
 
     """
     return db_to_raw_extrinsic_metadata(metadata)
 
 
 def revision_converter(db: BaseDb, revision_d: Dict[str, Any]) -> Revision:
     """Convert revision from the flat representation to swh model
        compatible objects.
 
     """
     revision = db_to_revision(revision_d)
     assert revision is not None, revision_d["id"]
     return revision
 
 
 def release_converter(db: BaseDb, release_d: Dict[str, Any]) -> Release:
     """Convert release from the flat representation to swh model
        compatible objects.
 
     """
     release = db_to_release(release_d)
     assert release is not None, release_d["id"]
     return release
 
 
 def snapshot_converter(db: BaseDb, snapshot_d: Dict[str, Any]) -> Snapshot:
     """Convert snapshot from the flat representation to swh model
        compatible objects.
 
     """
     columns = ["name", "target", "target_type"]
     query = """
     select %s
     from snapshot_branches sbs
     inner join snapshot_branch sb on sb.object_id=sbs.branch_id
     where sbs.snapshot_id=%%s
     """ % ", ".join(
         columns
     )
     with db.cursor() as cur:
         cur.execute(query, (snapshot_d["object_id"],))
         branches = {}
         for name, *row in cur:
             branch_d = dict(zip(columns[1:], row))
             if branch_d["target"] is not None and branch_d["target_type"] is not None:
                 branch: Optional[SnapshotBranch] = SnapshotBranch(
                     target=branch_d["target"],
                     target_type=TargetType(branch_d["target_type"]),
                 )
             else:
                 branch = None
             branches[name] = branch
 
     return Snapshot(id=snapshot_d["id"], branches=branches,)
 
 
 CONVERTERS: Dict[str, Callable[[BaseDb, Dict[str, Any]], BaseModel]] = {
     "directory": directory_converter,
     "raw_extrinsic_metadata": raw_extrinsic_metadata_converter,
     "revision": revision_converter,
     "release": release_converter,
     "snapshot": snapshot_converter,
 }
 
 
 def object_to_offset(object_id, numbits):
     """Compute the index of the range containing object id, when dividing
        space into 2^numbits.
 
     Args:
         object_id (str): The hex representation of object_id
         numbits (int): Number of bits in which we divide input space
 
     Returns:
         The index of the range containing object id
 
     """
     q, r = divmod(numbits, 8)
     length = q + (r != 0)
     shift_bits = 8 - r if r else 0
 
     truncated_id = object_id[: length * 2]
     if len(truncated_id) < length * 2:
         truncated_id += "0" * (length * 2 - len(truncated_id))
 
     truncated_id_bytes = bytes.fromhex(truncated_id)
     return int.from_bytes(truncated_id_bytes, byteorder="big") >> shift_bits
 
 
 def byte_ranges(numbits, start_object=None, end_object=None):
     """Generate start/end pairs of bytes spanning numbits bits and
        constrained by optional start_object and end_object.
 
     Args:
         numbits (int): Number of bits in which we divide input space
         start_object (str): Hex object id contained in the first range
                             returned
         end_object (str): Hex object id contained in the last range
                           returned
 
     Yields:
         2^numbits pairs of bytes
 
     """
     q, r = divmod(numbits, 8)
     length = q + (r != 0)
     shift_bits = 8 - r if r else 0
 
     def to_bytes(i):
         return int.to_bytes(i << shift_bits, length=length, byteorder="big")
 
     start_offset = 0
     end_offset = 1 << numbits
 
     if start_object is not None:
         start_offset = object_to_offset(start_object, numbits)
     if end_object is not None:
         end_offset = object_to_offset(end_object, numbits) + 1
 
     for start in range(start_offset, end_offset):
         end = start + 1
 
         if start == 0:
             yield None, to_bytes(end)
         elif end == 1 << numbits:
             yield to_bytes(start), None
         else:
             yield to_bytes(start), to_bytes(end)
 
 
 def integer_ranges(start, end, block_size=1000):
     for start in range(start, end, block_size):
         if start == 0:
             yield None, block_size
         elif start + block_size > end:
             yield start, end
         else:
             yield start, start + block_size
 
 
 RANGE_GENERATORS = {
     "content": lambda start, end: byte_ranges(24, start, end),
     "skipped_content": lambda start, end: [(None, None)],
     "directory": lambda start, end: byte_ranges(24, start, end),
     "revision": lambda start, end: byte_ranges(24, start, end),
     "release": lambda start, end: byte_ranges(16, start, end),
     "snapshot": lambda start, end: byte_ranges(16, start, end),
     "origin": integer_ranges,
     "origin_visit": integer_ranges,
     "origin_visit_status": integer_ranges,
 }
 
 
 def compute_query(obj_type, start, end):
     columns = COLUMNS.get(obj_type)
     join_specs = JOINS.get(obj_type, [])
     join_clause = "\n".join("left join %s" % clause for clause in join_specs)
 
     where = []
     where_args = []
     if start:
         where.append("%(keys)s >= %%s")
         where_args.append(start)
     if end:
         where.append("%(keys)s < %%s")
         where_args.append(end)
 
     where_clause = ""
     if where:
         where_clause = ("where " + " and ".join(where)) % {
             "keys": "(%s)" % PARTITION_KEY[obj_type]
         }
 
     column_specs = []
     column_aliases = []
     for column in columns:
         if isinstance(column, str):
             column_specs.append(column)
             column_aliases.append(column)
         else:
             column_specs.append("%s as %s" % column)
             column_aliases.append(column[1])
 
     query = """
 select %(columns)s
 from %(table)s
 %(join)s
 %(where)s
     """ % {
         "columns": ",".join(column_specs),
         "table": obj_type,
         "join": join_clause,
         "where": where_clause,
     }
 
     return query, where_args, column_aliases
 
 
 def fetch(db, obj_type, start, end):
     """Fetch all obj_type's identifiers from db.
 
     This opens one connection, stream objects and when done, close
     the connection.
 
     Args:
         db (BaseDb): Db connection object
         obj_type (str): Object type
         start (Union[bytes|Tuple]): Range start identifier
         end (Union[bytes|Tuple]): Range end identifier
 
     Raises:
         ValueError if obj_type is not supported
 
     Yields:
         Objects in the given range
 
     """
     query, where_args, column_aliases = compute_query(obj_type, start, end)
     converter = CONVERTERS.get(obj_type)
     with db.cursor() as cursor:
         logger.debug("Fetching data for table %s", obj_type)
         logger.debug("query: %s %s", query, where_args)
         cursor.execute(query, where_args)
         for row in cursor:
             record = dict(zip(column_aliases, row))
             if converter:
                 record = converter(db, record)
             else:
                 record = object_converter_fn[obj_type](record)
 
             logger.debug("record: %s", record)
             yield record
 
 
 def _format_range_bound(bound):
     if isinstance(bound, bytes):
         return bound.hex()
     else:
         return str(bound)
 
 
 MANDATORY_KEYS = ["storage", "journal_writer"]
 
 
 class JournalBackfiller:
     """Class in charge of reading the storage's objects and sends those
        back to the journal's topics.
 
        This is designed to be run periodically.
 
     """
 
     def __init__(self, config=None):
         self.config = config
         self.check_config(config)
 
     def check_config(self, config):
         missing_keys = []
         for key in MANDATORY_KEYS:
             if not config.get(key):
                 missing_keys.append(key)
 
         if missing_keys:
             raise ValueError(
                 "Configuration error: The following keys must be"
                 " provided: %s" % (",".join(missing_keys),)
             )
 
         if "cls" not in config["storage"] or config["storage"]["cls"] != "local":
             raise ValueError(
                 "swh storage backfiller must be configured to use a local"
                 " (PostgreSQL) storage"
             )
 
     def parse_arguments(self, object_type, start_object, end_object):
         """Parse arguments
 
         Raises:
             ValueError for unsupported object type
             ValueError if object ids are not parseable
 
         Returns:
             Parsed start and end object ids
 
         """
         if object_type not in COLUMNS:
             raise ValueError(
                 "Object type %s is not supported. "
                 "The only possible values are %s"
                 % (object_type, ", ".join(COLUMNS.keys()))
             )
 
-        if object_type in ["origin", "origin_visit"]:
+        if object_type in ["origin", "origin_visit", "origin_visit_status"]:
             if start_object:
                 start_object = int(start_object)
             else:
                 start_object = 0
             if end_object:
                 end_object = int(end_object)
             else:
                 end_object = 100 * 1000 * 1000  # hard-coded limit
 
         return start_object, end_object
 
     def run(self, object_type, start_object, end_object, dry_run=False):
         """Reads storage's subscribed object types and send them to the
            journal's reading topic.
 
         """
         start_object, end_object = self.parse_arguments(
             object_type, start_object, end_object
         )
 
         db = BaseDb.connect(self.config["storage"]["db"])
         writer = JournalWriter({"cls": "kafka", **self.config["journal_writer"]})
         assert writer.journal is not None
 
         for range_start, range_end in RANGE_GENERATORS[object_type](
             start_object, end_object
         ):
             logger.info(
                 "Processing %s range %s to %s",
                 object_type,
                 _format_range_bound(range_start),
                 _format_range_bound(range_end),
             )
 
             objects = fetch(db, object_type, start=range_start, end=range_end)
 
             if not dry_run:
                 writer.write_additions(object_type, objects)
             else:
                 # only consume the objects iterator to check for any potential
                 # decoding/encoding errors
                 for obj in objects:
                     pass
 
 
 if __name__ == "__main__":
     print('Please use the "swh-journal backfiller run" command')
diff --git a/swh/storage/cassandra/model.py b/swh/storage/cassandra/model.py
index 740a8802..2025df4b 100644
--- a/swh/storage/cassandra/model.py
+++ b/swh/storage/cassandra/model.py
@@ -1,277 +1,283 @@
 # Copyright (C) 2020  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 """Classes representing tables in the Cassandra database.
 
 They are very close to classes found in swh.model.model, but most of
 them are subtly different:
 
 * Large objects are split into other classes (eg. RevisionRow has no
   'parents' field, because parents are stored in a different table,
   represented by RevisionParentRow)
 * They have a "cols" field, which returns the list of column names
   of the table
 * They only use types that map directly to Cassandra's schema (ie. no enums)
 
 Therefore, this model doesn't reuse swh.model.model, except for types
 that can be mapped to UDTs (Person and TimestampWithTimezone).
 """
 
 import dataclasses
 import datetime
 from typing import Any, ClassVar, Dict, List, Optional, Tuple, Type, TypeVar
 
 from swh.model.model import Person, TimestampWithTimezone
 
 MAGIC_NULL_PK = b"<null>"
 """
 NULLs (or all-empty blobs) are not allowed in primary keys; instead we use a
 special value that can't possibly be a valid hash.
 """
 
 
 T = TypeVar("T", bound="BaseRow")
 
 
 class BaseRow:
     TABLE: ClassVar[str]
     PARTITION_KEY: ClassVar[Tuple[str, ...]]
     CLUSTERING_KEY: ClassVar[Tuple[str, ...]] = ()
 
     @classmethod
     def from_dict(cls: Type[T], d: Dict[str, Any]) -> T:
         return cls(**d)  # type: ignore
 
     @classmethod
     def cols(cls) -> List[str]:
         return [field.name for field in dataclasses.fields(cls)]
 
     def to_dict(self) -> Dict[str, Any]:
         return dataclasses.asdict(self)
 
 
 @dataclasses.dataclass
 class ContentRow(BaseRow):
     TABLE = "content"
     PARTITION_KEY = ("sha1", "sha1_git", "sha256", "blake2s256")
 
     sha1: bytes
     sha1_git: bytes
     sha256: bytes
     blake2s256: bytes
     length: int
     ctime: datetime.datetime
     status: str
 
 
 @dataclasses.dataclass
 class SkippedContentRow(BaseRow):
     TABLE = "skipped_content"
     PARTITION_KEY = ("sha1", "sha1_git", "sha256", "blake2s256")
 
     sha1: Optional[bytes]
     sha1_git: Optional[bytes]
     sha256: Optional[bytes]
     blake2s256: Optional[bytes]
     length: Optional[int]
     ctime: Optional[datetime.datetime]
     status: str
     reason: str
     origin: str
 
     @classmethod
     def from_dict(cls, d: Dict[str, Any]) -> "SkippedContentRow":
         d = d.copy()
         for k in ("sha1", "sha1_git", "sha256", "blake2s256"):
             if d[k] == MAGIC_NULL_PK:
                 d[k] = None
         return super().from_dict(d)
 
 
 @dataclasses.dataclass
 class DirectoryRow(BaseRow):
     TABLE = "directory"
     PARTITION_KEY = ("id",)
 
     id: bytes
 
 
 @dataclasses.dataclass
 class DirectoryEntryRow(BaseRow):
     TABLE = "directory_entry"
     PARTITION_KEY = ("directory_id",)
     CLUSTERING_KEY = ("name",)
 
     directory_id: bytes
     name: bytes
     target: bytes
     perms: int
     type: str
 
 
 @dataclasses.dataclass
 class RevisionRow(BaseRow):
     TABLE = "revision"
     PARTITION_KEY = ("id",)
 
     id: bytes
     date: Optional[TimestampWithTimezone]
     committer_date: Optional[TimestampWithTimezone]
     type: str
     directory: bytes
     message: bytes
     author: Person
     committer: Person
     synthetic: bool
     metadata: str
     extra_headers: dict
 
 
 @dataclasses.dataclass
 class RevisionParentRow(BaseRow):
     TABLE = "revision_parent"
     PARTITION_KEY = ("id",)
     CLUSTERING_KEY = ("parent_rank",)
 
     id: bytes
     parent_rank: int
     parent_id: bytes
 
 
 @dataclasses.dataclass
 class ReleaseRow(BaseRow):
     TABLE = "release"
     PARTITION_KEY = ("id",)
 
     id: bytes
     target_type: str
     target: bytes
     date: TimestampWithTimezone
     name: bytes
     message: bytes
     author: Person
     synthetic: bool
 
 
 @dataclasses.dataclass
 class SnapshotRow(BaseRow):
     TABLE = "snapshot"
     PARTITION_KEY = ("id",)
 
     id: bytes
 
 
 @dataclasses.dataclass
 class SnapshotBranchRow(BaseRow):
     TABLE = "snapshot_branch"
     PARTITION_KEY = ("snapshot_id",)
     CLUSTERING_KEY = ("name",)
 
     snapshot_id: bytes
     name: bytes
     target_type: Optional[str]
     target: Optional[bytes]
 
 
 @dataclasses.dataclass
 class OriginVisitRow(BaseRow):
     TABLE = "origin_visit"
     PARTITION_KEY = ("origin",)
     CLUSTERING_KEY = ("visit",)
 
     origin: str
     visit: int
     date: datetime.datetime
     type: str
 
 
 @dataclasses.dataclass
 class OriginVisitStatusRow(BaseRow):
     TABLE = "origin_visit_status"
     PARTITION_KEY = ("origin",)
     CLUSTERING_KEY = ("visit", "date")
 
     origin: str
     visit: int
     date: datetime.datetime
     status: str
     metadata: str
     snapshot: bytes
 
+    @classmethod
+    def from_dict(cls: Type[T], d: Dict[str, Any]) -> T:
+        d = d.copy()
+        d.pop("type", None)
+        return cls(**d)  # type: ignore
+
 
 @dataclasses.dataclass
 class OriginRow(BaseRow):
     TABLE = "origin"
     PARTITION_KEY = ("sha1",)
 
     sha1: bytes
     url: str
     next_visit_id: int
 
 
 @dataclasses.dataclass
 class MetadataAuthorityRow(BaseRow):
     TABLE = "metadata_authority"
     PARTITION_KEY = ("url",)
     CLUSTERING_KEY = ("type",)
 
     url: str
     type: str
     metadata: str
 
 
 @dataclasses.dataclass
 class MetadataFetcherRow(BaseRow):
     TABLE = "metadata_fetcher"
     PARTITION_KEY = ("name",)
     CLUSTERING_KEY = ("version",)
 
     name: str
     version: str
     metadata: str
 
 
 @dataclasses.dataclass
 class RawExtrinsicMetadataRow(BaseRow):
     TABLE = "raw_extrinsic_metadata"
     PARTITION_KEY = ("target",)
     CLUSTERING_KEY = (
         "authority_type",
         "authority_url",
         "discovery_date",
         "fetcher_name",
         "fetcher_version",
     )
 
     type: str
     target: str
 
     authority_type: str
     authority_url: str
     discovery_date: datetime.datetime
     fetcher_name: str
     fetcher_version: str
 
     format: str
     metadata: bytes
 
     origin: Optional[str]
     visit: Optional[int]
     snapshot: Optional[str]
     release: Optional[str]
     revision: Optional[str]
     path: Optional[bytes]
     directory: Optional[str]
 
 
 @dataclasses.dataclass
 class ObjectCountRow(BaseRow):
     TABLE = "object_count"
     PARTITION_KEY = ("partition_key",)
     CLUSTERING_KEY = ("object_type",)
 
     partition_key: int
     object_type: str
     count: int
diff --git a/swh/storage/cassandra/schema.py b/swh/storage/cassandra/schema.py
index 9c099d6c..45a1d925 100644
--- a/swh/storage/cassandra/schema.py
+++ b/swh/storage/cassandra/schema.py
@@ -1,282 +1,272 @@
-# Copyright (C) 2019-2020  The Software Heritage developers
+# Copyright (C) 2019-2021  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 
-CREATE_TABLES_QUERIES = """
+CREATE_TABLES_QUERIES = [
+    """
 CREATE OR REPLACE FUNCTION ascii_bins_count_sfunc (
     state tuple<int, map<ascii, int>>, -- (nb_none, map<target_type, nb>)
     bin_name ascii
 )
 CALLED ON NULL INPUT
 RETURNS tuple<int, map<ascii, int>>
 LANGUAGE java AS
 $$
     if (bin_name == null) {
         state.setInt(0, state.getInt(0) + 1);
     }
     else {
         Map<String, Integer> counters = state.getMap(
             1, String.class, Integer.class);
         Integer nb = counters.get(bin_name);
         if (nb == null) {
             nb = 0;
         }
         counters.put(bin_name, nb + 1);
         state.setMap(1, counters, String.class, Integer.class);
     }
     return state;
 $$
-;
-
-
+;""",
+    """
 CREATE OR REPLACE AGGREGATE ascii_bins_count ( ascii )
 SFUNC ascii_bins_count_sfunc
 STYPE tuple<int, map<ascii, int>>
 INITCOND (0, {})
-;
-
-
+;""",
+    """
 CREATE TYPE IF NOT EXISTS microtimestamp (
     seconds             bigint,
     microseconds        int
-);
-
-
+);""",
+    """
 CREATE TYPE IF NOT EXISTS microtimestamp_with_timezone (
     timestamp           frozen<microtimestamp>,
     offset              smallint,
     negative_utc        boolean
-);
-
-
+);""",
+    """
 CREATE TYPE IF NOT EXISTS person (
     fullname    blob,
     name        blob,
     email       blob
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS content (
     sha1          blob,
     sha1_git      blob,
     sha256        blob,
     blake2s256    blob,
     length        bigint,
     ctime         timestamp,
         -- creation time, i.e. time of (first) injection into the storage
     status        ascii,
     PRIMARY KEY ((sha1, sha1_git, sha256, blake2s256))
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS skipped_content (
     sha1          blob,
     sha1_git      blob,
     sha256        blob,
     blake2s256    blob,
     length        bigint,
     ctime         timestamp,
         -- creation time, i.e. time of (first) injection into the storage
     status        ascii,
     reason        text,
     origin        text,
     PRIMARY KEY ((sha1, sha1_git, sha256, blake2s256))
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS revision (
     id                              blob PRIMARY KEY,
     date                            microtimestamp_with_timezone,
     committer_date                  microtimestamp_with_timezone,
     type                            ascii,
     directory                       blob,  -- source code "root" directory
     message                         blob,
     author                          person,
     committer                       person,
     synthetic                       boolean,
         -- true iff revision has been created by Software Heritage
     metadata                        text,
         -- extra metadata as JSON(tarball checksums, etc...)
     extra_headers                   frozen<list <list<blob>> >
         -- extra commit information as (tuple(key, value), ...)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS revision_parent (
     id                     blob,
     parent_rank                     int,
         -- parent position in merge commits, 0-based
     parent_id                       blob,
     PRIMARY KEY ((id), parent_rank)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS release
 (
     id                              blob PRIMARY KEY,
     target_type                     ascii,
     target                          blob,
     date                            microtimestamp_with_timezone,
     name                            blob,
     message                         blob,
     author                          person,
     synthetic                       boolean,
         -- true iff release has been created by Software Heritage
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS directory (
     id              blob PRIMARY KEY,
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS directory_entry (
     directory_id    blob,
     name            blob,  -- path name, relative to containing dir
     target          blob,
     perms           int,   -- unix-like permissions
     type            ascii, -- target type
     PRIMARY KEY ((directory_id), name)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS snapshot (
     id              blob PRIMARY KEY,
-);
-
-
+);""",
+    """
 -- For a given snapshot_id, branches are sorted by their name,
 -- allowing easy pagination.
 CREATE TABLE IF NOT EXISTS snapshot_branch (
     snapshot_id     blob,
     name            blob,
     target_type     ascii,
     target          blob,
     PRIMARY KEY ((snapshot_id), name)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS origin_visit (
     origin          text,
     visit           bigint,
     date            timestamp,
     type            text,
     PRIMARY KEY ((origin), visit)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS origin_visit_status (
     origin          text,
     visit           bigint,
     date            timestamp,
     status          ascii,
     metadata        text,
     snapshot        blob,
     PRIMARY KEY ((origin), visit, date)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS origin (
     sha1            blob PRIMARY KEY,
     url             text,
     next_visit_id   int,
         -- We need integer visit ids for compatibility with the pgsql
         -- storage, so we're using lightweight transactions with this trick:
         -- https://stackoverflow.com/a/29391877/539465
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS metadata_authority (
     url             text,
     type            ascii,
     metadata        text,
     PRIMARY KEY ((url), type)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS metadata_fetcher (
     name            ascii,
     version         ascii,
     metadata        text,
     PRIMARY KEY ((name), version)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS raw_extrinsic_metadata (
     type            text,
     target          text,
 
     -- metadata source
     authority_type  text,
     authority_url   text,
     discovery_date  timestamp,
     fetcher_name    ascii,
     fetcher_version ascii,
 
     -- metadata itself
     format          ascii,
     metadata        blob,
 
     -- context
     origin          text,
     visit           bigint,
     snapshot        text,
     release         text,
     revision        text,
     path            blob,
     directory       text,
 
     PRIMARY KEY ((target), authority_type, authority_url, discovery_date,
                            fetcher_name, fetcher_version)
-);
-
-
+);""",
+    """
 CREATE TABLE IF NOT EXISTS object_count (
     partition_key   smallint,  -- Constant, must always be 0
     object_type     ascii,
     count           counter,
     PRIMARY KEY ((partition_key), object_type)
-);
-""".split(
-    "\n\n\n"
-)
+);""",
+]
 
 CONTENT_INDEX_TEMPLATE = """
 -- Secondary table, used for looking up "content" from a single hash
 CREATE TABLE IF NOT EXISTS content_by_{main_algo} (
     {main_algo}   blob,
     target_token  bigint, -- value of token(pk) on the "primary" table
     PRIMARY KEY (({main_algo}), target_token)
 );
 
 CREATE TABLE IF NOT EXISTS skipped_content_by_{main_algo} (
     {main_algo}   blob,
     target_token  bigint, -- value of token(pk) on the "primary" table
     PRIMARY KEY (({main_algo}), target_token)
 );
 """
 
-TABLES = (
-    "skipped_content content revision revision_parent release "
-    "directory directory_entry snapshot snapshot_branch "
-    "origin_visit origin raw_extrinsic_metadata object_count "
-    "origin_visit_status metadata_authority "
-    "metadata_fetcher"
-).split()
+TABLES = [
+    "skipped_content",
+    "content",
+    "revision",
+    "revision_parent",
+    "release",
+    "directory",
+    "directory_entry",
+    "snapshot",
+    "snapshot_branch",
+    "origin_visit",
+    "origin",
+    "raw_extrinsic_metadata",
+    "object_count",
+    "origin_visit_status",
+    "metadata_authority",
+    "metadata_fetcher",
+]
 
 HASH_ALGORITHMS = ["sha1", "sha1_git", "sha256", "blake2s256"]
 
 for main_algo in HASH_ALGORITHMS:
     CREATE_TABLES_QUERIES.extend(
         CONTENT_INDEX_TEMPLATE.format(
             main_algo=main_algo,
             other_algos=", ".join(
                 [algo for algo in HASH_ALGORITHMS if algo != main_algo]
             ),
         ).split("\n\n")
     )
 
     TABLES.append("content_by_%s" % main_algo)
     TABLES.append("skipped_content_by_%s" % main_algo)
diff --git a/swh/storage/tests/test_cassandra.py b/swh/storage/tests/test_cassandra.py
index 4ec1ffa3..bdadb9e1 100644
--- a/swh/storage/tests/test_cassandra.py
+++ b/swh/storage/tests/test_cassandra.py
@@ -1,415 +1,419 @@
-# Copyright (C) 2018-2020  The Software Heritage developers
+# Copyright (C) 2018-2021  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import datetime
 import os
 import signal
 import socket
 import subprocess
 import time
 from typing import Dict
 
 import attr
 import pytest
 
 from swh.core.api.classes import stream_results
 from swh.storage import get_storage
 from swh.storage.cassandra import create_keyspace
 from swh.storage.cassandra.model import ContentRow
 from swh.storage.cassandra.schema import HASH_ALGORITHMS, TABLES
 from swh.storage.tests.storage_tests import (
     TestStorageGeneratedData as _TestStorageGeneratedData,
 )
 from swh.storage.tests.storage_tests import TestStorage as _TestStorage
 from swh.storage.utils import now
 
 CONFIG_TEMPLATE = """
 data_file_directories:
     - {data_dir}/data
 commitlog_directory: {data_dir}/commitlog
 hints_directory: {data_dir}/hints
 saved_caches_directory: {data_dir}/saved_caches
 
 commitlog_sync: periodic
 commitlog_sync_period_in_ms: 1000000
 partitioner: org.apache.cassandra.dht.Murmur3Partitioner
 endpoint_snitch: SimpleSnitch
 seed_provider:
     - class_name: org.apache.cassandra.locator.SimpleSeedProvider
       parameters:
           - seeds: "127.0.0.1"
 
 storage_port: {storage_port}
 native_transport_port: {native_transport_port}
 start_native_transport: true
 listen_address: 127.0.0.1
 
 enable_user_defined_functions: true
 
 # speed-up by disabling period saving to disk
 key_cache_save_period: 0
 row_cache_save_period: 0
 trickle_fsync: false
 commitlog_sync_period_in_ms: 100000
 """
 
 
 def free_port():
     sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
     sock.bind(("127.0.0.1", 0))
     port = sock.getsockname()[1]
     sock.close()
     return port
 
 
 def wait_for_peer(addr, port):
     wait_until = time.time() + 20
     while time.time() < wait_until:
         try:
             sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
             sock.connect((addr, port))
         except ConnectionRefusedError:
             time.sleep(0.1)
         else:
             sock.close()
             return True
     return False
 
 
 @pytest.fixture(scope="session")
 def cassandra_cluster(tmpdir_factory):
     cassandra_conf = tmpdir_factory.mktemp("cassandra_conf")
     cassandra_data = tmpdir_factory.mktemp("cassandra_data")
     cassandra_log = tmpdir_factory.mktemp("cassandra_log")
     native_transport_port = free_port()
     storage_port = free_port()
     jmx_port = free_port()
 
     with open(str(cassandra_conf.join("cassandra.yaml")), "w") as fd:
         fd.write(
             CONFIG_TEMPLATE.format(
                 data_dir=str(cassandra_data),
                 storage_port=storage_port,
                 native_transport_port=native_transport_port,
             )
         )
 
     if os.environ.get("SWH_CASSANDRA_LOG"):
         stdout = stderr = None
     else:
         stdout = stderr = subprocess.DEVNULL
 
     cassandra_bin = os.environ.get("SWH_CASSANDRA_BIN", "/usr/sbin/cassandra")
+    env = {
+        "MAX_HEAP_SIZE": "300M",
+        "HEAP_NEWSIZE": "50M",
+        "JVM_OPTS": "-Xlog:gc=error:file=%s/gc.log" % cassandra_log,
+    }
+    if "JAVA_HOME" in os.environ:
+        env["JAVA_HOME"] = os.environ["JAVA_HOME"]
+
     proc = subprocess.Popen(
         [
             cassandra_bin,
             "-Dcassandra.config=file://%s/cassandra.yaml" % cassandra_conf,
             "-Dcassandra.logdir=%s" % cassandra_log,
             "-Dcassandra.jmx.local.port=%d" % jmx_port,
             "-Dcassandra-foreground=yes",
         ],
         start_new_session=True,
-        env={
-            "MAX_HEAP_SIZE": "300M",
-            "HEAP_NEWSIZE": "50M",
-            "JVM_OPTS": "-Xlog:gc=error:file=%s/gc.log" % cassandra_log,
-        },
+        env=env,
         stdout=stdout,
         stderr=stderr,
     )
 
     running = wait_for_peer("127.0.0.1", native_transport_port)
 
     if running:
         yield (["127.0.0.1"], native_transport_port)
 
     if not running or os.environ.get("SWH_CASSANDRA_LOG"):
         debug_log_path = str(cassandra_log.join("debug.log"))
         if os.path.exists(debug_log_path):
             with open(debug_log_path) as fd:
                 print(fd.read())
 
     if not running:
         raise Exception("cassandra process stopped unexpectedly.")
 
     pgrp = os.getpgid(proc.pid)
     os.killpg(pgrp, signal.SIGKILL)
 
 
 class RequestHandler:
     def on_request(self, rf):
         if hasattr(rf.message, "query"):
             print()
             print(rf.message.query)
 
 
 @pytest.fixture(scope="session")
 def keyspace(cassandra_cluster):
     (hosts, port) = cassandra_cluster
     keyspace = os.urandom(10).hex()
 
     create_keyspace(hosts, keyspace, port)
 
     return keyspace
 
 
 # tests are executed using imported classes (TestStorage and
 # TestStorageGeneratedData) using overloaded swh_storage fixture
 # below
 
 
 @pytest.fixture
 def swh_storage_backend_config(cassandra_cluster, keyspace):
     (hosts, port) = cassandra_cluster
 
     storage_config = dict(
         cls="cassandra",
         hosts=hosts,
         port=port,
         keyspace=keyspace,
         journal_writer={"cls": "memory"},
         objstorage={"cls": "memory"},
     )
 
     yield storage_config
 
     storage = get_storage(**storage_config)
 
     for table in TABLES:
         storage._cql_runner._session.execute('TRUNCATE TABLE "%s"' % table)
 
     storage._cql_runner._cluster.shutdown()
 
 
 @pytest.mark.cassandra
 class TestCassandraStorage(_TestStorage):
     def test_content_add_murmur3_collision(self, swh_storage, mocker, sample_data):
         """The Murmur3 token is used as link from index tables to the main
         table; and non-matching contents with colliding murmur3-hash
         are filtered-out when reading the main table.
         This test checks the content methods do filter out these collision.
         """
         called = 0
 
         cont, cont2 = sample_data.contents[:2]
 
         # always return a token
         def mock_cgtfsh(algo, hash_):
             nonlocal called
             called += 1
             assert algo in ("sha1", "sha1_git")
             return [123456]
 
         mocker.patch.object(
             swh_storage._cql_runner, "content_get_tokens_from_single_hash", mock_cgtfsh,
         )
 
         # For all tokens, always return cont
         def mock_cgft(token):
             nonlocal called
             called += 1
             return [
                 ContentRow(
                     length=10,
                     ctime=datetime.datetime.now(),
                     status="present",
                     **{algo: getattr(cont, algo) for algo in HASH_ALGORITHMS},
                 )
             ]
 
         mocker.patch.object(
             swh_storage._cql_runner, "content_get_from_token", mock_cgft
         )
 
         actual_result = swh_storage.content_add([cont2])
 
         assert called == 4
         assert actual_result == {
             "content:add": 1,
             "content:add:bytes": cont2.length,
         }
 
     def test_content_get_metadata_murmur3_collision(
         self, swh_storage, mocker, sample_data
     ):
         """The Murmur3 token is used as link from index tables to the main
         table; and non-matching contents with colliding murmur3-hash
         are filtered-out when reading the main table.
         This test checks the content methods do filter out these collisions.
         """
         called = 0
 
         cont, cont2 = [attr.evolve(c, ctime=now()) for c in sample_data.contents[:2]]
 
         # always return a token
         def mock_cgtfsh(algo, hash_):
             nonlocal called
             called += 1
             assert algo in ("sha1", "sha1_git")
             return [123456]
 
         mocker.patch.object(
             swh_storage._cql_runner, "content_get_tokens_from_single_hash", mock_cgtfsh,
         )
 
         # For all tokens, always return cont and cont2
         cols = list(set(cont.to_dict()) - {"data"})
 
         def mock_cgft(token):
             nonlocal called
             called += 1
             return [
                 ContentRow(**{col: getattr(cont, col) for col in cols},)
                 for cont in [cont, cont2]
             ]
 
         mocker.patch.object(
             swh_storage._cql_runner, "content_get_from_token", mock_cgft
         )
 
         actual_result = swh_storage.content_get([cont.sha1])
         assert called == 2
 
         # dropping extra column not returned
         expected_cont = attr.evolve(cont, data=None)
 
         # but cont2 should be filtered out
         assert actual_result == [expected_cont]
 
     def test_content_find_murmur3_collision(self, swh_storage, mocker, sample_data):
         """The Murmur3 token is used as link from index tables to the main
         table; and non-matching contents with colliding murmur3-hash
         are filtered-out when reading the main table.
         This test checks the content methods do filter out these collisions.
         """
         called = 0
 
         cont, cont2 = [attr.evolve(c, ctime=now()) for c in sample_data.contents[:2]]
 
         # always return a token
         def mock_cgtfsh(algo, hash_):
             nonlocal called
             called += 1
             assert algo in ("sha1", "sha1_git")
             return [123456]
 
         mocker.patch.object(
             swh_storage._cql_runner, "content_get_tokens_from_single_hash", mock_cgtfsh,
         )
 
         # For all tokens, always return cont and cont2
         cols = list(set(cont.to_dict()) - {"data"})
 
         def mock_cgft(token):
             nonlocal called
             called += 1
             return [
                 ContentRow(**{col: getattr(cont, col) for col in cols})
                 for cont in [cont, cont2]
             ]
 
         mocker.patch.object(
             swh_storage._cql_runner, "content_get_from_token", mock_cgft
         )
 
         expected_content = attr.evolve(cont, data=None)
 
         actual_result = swh_storage.content_find({"sha1": cont.sha1})
 
         assert called == 2
 
         # but cont2 should be filtered out
         assert actual_result == [expected_content]
 
     def test_content_get_partition_murmur3_collision(
         self, swh_storage, mocker, sample_data
     ):
         """The Murmur3 token is used as link from index tables to the main table; and
         non-matching contents with colliding murmur3-hash are filtered-out when reading
         the main table.
 
         This test checks the content_get_partition endpoints return all contents, even
         the collisions.
 
         """
         called = 0
 
         rows: Dict[int, Dict] = {}
         for tok, content in enumerate(sample_data.contents):
             cont = attr.evolve(content, data=None, ctime=now())
             row_d = {**cont.to_dict(), "tok": tok}
             rows[tok] = row_d
 
         # For all tokens, always return cont
 
         def mock_content_get_token_range(range_start, range_end, limit):
             nonlocal called
             called += 1
 
             for tok in list(rows.keys()) * 3:  # yield multiple times the same tok
                 row_d = dict(rows[tok].items())
                 row_d.pop("tok")
                 yield (tok, ContentRow(**row_d))
 
         mocker.patch.object(
             swh_storage._cql_runner,
             "content_get_token_range",
             mock_content_get_token_range,
         )
 
         actual_results = list(
             stream_results(
                 swh_storage.content_get_partition, partition_id=0, nb_partitions=1
             )
         )
 
         assert called > 0
 
         # everything is listed, even collisions
         assert len(actual_results) == 3 * len(sample_data.contents)
         # as we duplicated the returned results, dropping duplicate should yield
         # the original length
         assert len(set(actual_results)) == len(sample_data.contents)
 
     @pytest.mark.skip("content_update is not yet implemented for Cassandra")
     def test_content_update(self):
         pass
 
     @pytest.mark.skip(
         'The "person" table of the pgsql is a legacy thing, and not '
         "supported by the cassandra backend."
     )
     def test_person_fullname_unicity(self):
         pass
 
     @pytest.mark.skip(
         'The "person" table of the pgsql is a legacy thing, and not '
         "supported by the cassandra backend."
     )
     def test_person_get(self):
         pass
 
     @pytest.mark.skip("Not supported by Cassandra")
     def test_origin_count(self):
         pass
 
 
 @pytest.mark.cassandra
 class TestCassandraStorageGeneratedData(_TestStorageGeneratedData):
     @pytest.mark.skip("Not supported by Cassandra")
     def test_origin_count(self):
         pass
 
     @pytest.mark.skip("Not supported by Cassandra")
     def test_origin_count_with_visit_no_visits(self):
         pass
 
     @pytest.mark.skip("Not supported by Cassandra")
     def test_origin_count_with_visit_with_visits_and_snapshot(self):
         pass
 
     @pytest.mark.skip("Not supported by Cassandra")
     def test_origin_count_with_visit_with_visits_no_snapshot(self):
         pass
diff --git a/tox.ini b/tox.ini
index 98b21039..1710b45f 100644
--- a/tox.ini
+++ b/tox.ini
@@ -1,42 +1,43 @@
 [tox]
 envlist=black,flake8,mypy,py3
 
 [testenv]
 extras =
   testing
 deps =
   pytest-cov
   dev: ipdb
 passenv =
   SWH_CASSANDRA_BIN
   SWH_CASSANDRA_LOG
+  JAVA_HOME
 commands =
   pytest \
     !slow: --hypothesis-profile=fast \
     slow:  --hypothesis-profile=slow \
          --cov={envsitepackagesdir}/swh/storage \
          {envsitepackagesdir}/swh/storage \
          --doctest-modules \
          --cov-branch {posargs}
 
 [testenv:black]
 skip_install = true
 deps =
   black==19.10b0
 commands =
   {envpython} -m black --check swh
 
 [testenv:flake8]
 skip_install = true
 deps =
   flake8
 commands =
   {envpython} -m flake8
 
 [testenv:mypy]
 extras =
   testing
 deps =
   mypy
 commands =
   mypy swh