diff --git a/.gitignore b/.gitignore
index 8aeb38a..1c279bb 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,15 +1,16 @@
 *.pyc
 *.sw?
 *~
 /.coverage
 /.coverage.*
 .eggs/
 __pycache__
 *.egg-info/
 build/
 dist/
 version.txt
 /sql/createdb-stamp
 /sql/filldb-stamp
 .tox/
-.hypothesis/
\ No newline at end of file
+.hypothesis/
+.mypy_cache/
diff --git a/MANIFEST.in b/MANIFEST.in
index 304d9f7..cfe6e5b 100644
--- a/MANIFEST.in
+++ b/MANIFEST.in
@@ -1,8 +1,9 @@
 include README.md
 include Makefile
 include requirements.txt
 include requirements-swh.txt
 include version.txt
 recursive-include sql *
 recursive-include swh/indexer/sql *.sql
 recursive-include swh/indexer/data *
+recursive-include swh py.typed
diff --git a/mypy.ini b/mypy.ini
new file mode 100644
index 0000000..08e40c5
--- /dev/null
+++ b/mypy.ini
@@ -0,0 +1,27 @@
+[mypy]
+namespace_packages = True
+warn_unused_ignores = True
+
+
+# 3rd party libraries without stubs (yet)
+
+[mypy-celery.*]
+ignore_missing_imports = True
+
+[mypy-magic.*]
+ignore_missing_imports = True
+
+[mypy-pkg_resources.*]
+ignore_missing_imports = True
+
+[mypy-psycopg2.*]
+ignore_missing_imports = True
+
+[mypy-pyld.*]
+ignore_missing_imports = True
+
+[mypy-pytest.*]
+ignore_missing_imports = True
+
+[mypy-xmltodict.*]
+ignore_missing_imports = True
diff --git a/swh/__init__.py b/swh/__init__.py
index 69e3be5..f14e196 100644
--- a/swh/__init__.py
+++ b/swh/__init__.py
@@ -1 +1,4 @@
-__path__ = __import__('pkgutil').extend_path(__path__, __name__)
+from pkgutil import extend_path
+from typing import Iterable
+
+__path__ = extend_path(__path__, __name__)  # type: Iterable[str]
diff --git a/swh/indexer/cli.py b/swh/indexer/cli.py
index ad94ba4..f1d5d79 100644
--- a/swh/indexer/cli.py
+++ b/swh/indexer/cli.py
@@ -1,259 +1,259 @@
 # Copyright (C) 2019  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import functools
 import json
 import time
 
 import click
 
 from swh.core import config
 from swh.core.cli import CONTEXT_SETTINGS, AliasedGroup
 from swh.journal.cli import get_journal_client
 from swh.scheduler import get_scheduler
 from swh.scheduler.cli_utils import schedule_origin_batches
 from swh.storage import get_storage
 
 from swh.indexer import metadata_dictionary
 from swh.indexer.journal_client import process_journal_objects
 from swh.indexer.storage import get_indexer_storage
 from swh.indexer.storage.api.server import load_and_check_config, app
 
 
 @click.group(name='indexer', context_settings=CONTEXT_SETTINGS,
              cls=AliasedGroup)
 @click.option('--config-file', '-C', default=None,
               type=click.Path(exists=True, dir_okay=False,),
               help="Configuration file.")
 @click.pass_context
 def cli(ctx, config_file):
     """Software Heritage Indexer tools.
 
     The Indexer is used to mine the content of the archive and extract derived
     information from archive source code artifacts.
 
     """
     ctx.ensure_object(dict)
 
     conf = config.read(config_file)
     ctx.obj['config'] = conf
 
 
 def _get_api(getter, config, config_key, url):
     if url:
         config[config_key] = {
             'cls': 'remote',
             'args': {'url': url}
         }
     elif config_key not in config:
         raise click.ClickException(
             'Missing configuration for {}'.format(config_key))
     return getter(**config[config_key])
 
 
 @cli.group('mapping')
 def mapping():
     '''Manage Software Heritage Indexer mappings.'''
     pass
 
 
 @mapping.command('list')
 def mapping_list():
     """Prints the list of known mappings."""
     mapping_names = [mapping.name
                      for mapping in metadata_dictionary.MAPPINGS.values()]
     mapping_names.sort()
     for mapping_name in mapping_names:
         click.echo(mapping_name)
 
 
 @mapping.command('list-terms')
 @click.option('--exclude-mapping', multiple=True,
               help='Exclude the given mapping from the output')
 @click.option('--concise', is_flag=True,
               default=False,
               help='Don\'t print the list of mappings supporting each term.')
 def mapping_list_terms(concise, exclude_mapping):
     """Prints the list of known CodeMeta terms, and which mappings
     support them."""
     properties = metadata_dictionary.list_terms()
     for (property_name, supported_mappings) in sorted(properties.items()):
         supported_mappings = {m.name for m in supported_mappings}
         supported_mappings -= set(exclude_mapping)
         if supported_mappings:
             if concise:
                 click.echo(property_name)
             else:
                 click.echo('{}:'.format(property_name))
                 click.echo('\t' + ', '.join(sorted(supported_mappings)))
 
 
 @mapping.command('translate')
 @click.argument('mapping-name')
 @click.argument('file', type=click.File('rb'))
 def mapping_translate(mapping_name, file):
     """Prints the list of known mappings."""
     mapping_cls = [cls for cls in metadata_dictionary.MAPPINGS.values()
                    if cls.name == mapping_name]
     if not mapping_cls:
         raise click.ClickException('Unknown mapping {}'.format(mapping_name))
     assert len(mapping_cls) == 1
     mapping_cls = mapping_cls[0]
     mapping = mapping_cls()
     codemeta_doc = mapping.translate(file.read())
     click.echo(json.dumps(codemeta_doc, indent=4))
 
 
 @cli.group('schedule')
 @click.option('--scheduler-url', '-s', default=None,
               help="URL of the scheduler API")
 @click.option('--indexer-storage-url', '-i', default=None,
               help="URL of the indexer storage API")
 @click.option('--storage-url', '-g', default=None,
               help="URL of the (graph) storage API")
 @click.option('--dry-run/--no-dry-run', is_flag=True,
               default=False,
               help='List only what would be scheduled.')
 @click.pass_context
 def schedule(ctx, scheduler_url, storage_url, indexer_storage_url,
              dry_run):
     """Manipulate Software Heritage Indexer tasks.
 
     Via SWH Scheduler's API."""
     ctx.obj['indexer_storage'] = _get_api(
         get_indexer_storage,
         ctx.obj['config'],
         'indexer_storage',
         indexer_storage_url
     )
     ctx.obj['storage'] = _get_api(
         get_storage,
         ctx.obj['config'],
         'storage',
         storage_url
     )
     ctx.obj['scheduler'] = _get_api(
         get_scheduler,
         ctx.obj['config'],
         'scheduler',
         scheduler_url
     )
     if dry_run:
         ctx.obj['scheduler'] = None
 
 
 def list_origins_by_producer(idx_storage, mappings, tool_ids):
     start = 0
     limit = 10000
     while True:
         origins = list(
             idx_storage.origin_intrinsic_metadata_search_by_producer(
                 start=start, limit=limit, ids_only=True,
                 mappings=mappings or None, tool_ids=tool_ids or None))
         if not origins:
             break
         start = origins[-1]+1
         yield from origins
 
 
 @schedule.command('reindex_origin_metadata')
 @click.option('--batch-size', '-b', 'origin_batch_size',
               default=10, show_default=True, type=int,
               help="Number of origins per task")
 @click.option('--tool-id', '-t', 'tool_ids', type=int, multiple=True,
               help="Restrict search of old metadata to this/these tool ids.")
 @click.option('--mapping', '-m', 'mappings', multiple=True,
               help="Mapping(s) that should be re-scheduled (eg. 'npm', "
                    "'gemspec', 'maven')")
 @click.option('--task-type',
               default='index-origin-metadata', show_default=True,
               help="Name of the task type to schedule.")
 @click.pass_context
 def schedule_origin_metadata_reindex(
         ctx, origin_batch_size, tool_ids, mappings, task_type):
     """Schedules indexing tasks for origins that were already indexed."""
     idx_storage = ctx.obj['indexer_storage']
     scheduler = ctx.obj['scheduler']
 
     origins = list_origins_by_producer(idx_storage, mappings, tool_ids)
 
     kwargs = {"policy_update": "update-dups"}
     schedule_origin_batches(
         scheduler, task_type, origins, origin_batch_size, kwargs)
 
 
 @cli.command('journal-client')
 @click.option('--scheduler-url', '-s', default=None,
               help="URL of the scheduler API")
 @click.option('--origin-metadata-task-type',
               default='index-origin-metadata',
               help='Name of the task running the origin metadata indexer.')
 @click.option('--broker', 'brokers', type=str, multiple=True,
               help='Kafka broker to connect to.')
 @click.option('--prefix', type=str, default=None,
               help='Prefix of Kafka topic names to read from.')
 @click.option('--group-id', type=str,
               help='Consumer/group id for reading from Kafka.')
 @click.option('--max-messages', '-m', default=None, type=int,
               help='Maximum number of objects to replay. Default is to '
                    'run forever.')
 @click.pass_context
 def journal_client(ctx, scheduler_url, origin_metadata_task_type,
                    brokers, prefix, group_id, max_messages):
     """Listens for new objects from the SWH Journal, and schedules tasks
     to run relevant indexers (currently, only origin-intrinsic-metadata)
     on these new objects."""
     scheduler = _get_api(
         get_scheduler,
         ctx.obj['config'],
         'scheduler',
         scheduler_url
     )
 
     client = get_journal_client(
         ctx, brokers=brokers, prefix=prefix, group_id=group_id,
         object_types=['origin_visit'])
 
     worker_fn = functools.partial(
         process_journal_objects,
         scheduler=scheduler,
         task_names={
             'origin_metadata': origin_metadata_task_type,
         }
     )
     nb_messages = 0
     last_log_time = 0
     try:
         while not max_messages or nb_messages < max_messages:
             nb_messages += client.process(worker_fn)
             if time.monotonic() - last_log_time >= 60:
                 print('Processed %d messages.' % nb_messages)
                 last_log_time = time.monotonic()
     except KeyboardInterrupt:
         ctx.exit(0)
     else:
         print('Done.')
 
 
 @cli.command('rpc-serve')
-@click.argument('config-path', required=1)
+@click.argument('config-path', required=True)
 @click.option('--host', default='0.0.0.0', help="Host to run the server")
 @click.option('--port', default=5007, type=click.INT,
               help="Binding port of the server")
 @click.option('--debug/--nodebug', default=True,
               help="Indicates if the server should run in debug mode")
 def rpc_server(config_path, host, port, debug):
     """Starts a Software Heritage Indexer RPC HTTP server."""
     api_cfg = load_and_check_config(config_path, type='any')
     app.config.update(api_cfg)
     app.run(host, port=int(port), debug=bool(debug))
 
 
 def main():
     return cli(auto_envvar_prefix='SWH_INDEXER')
 
 
 if __name__ == '__main__':
     main()
diff --git a/swh/indexer/fossology_license.py b/swh/indexer/fossology_license.py
index 017d918..ed9a431 100644
--- a/swh/indexer/fossology_license.py
+++ b/swh/indexer/fossology_license.py
@@ -1,172 +1,173 @@
 # Copyright (C) 2016-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import subprocess
 
-from swh.model import hashutil
+from typing import Optional
 
+from swh.model import hashutil
 from .indexer import ContentIndexer, ContentRangeIndexer, write_to_temp
 
 
 def compute_license(path, log=None):
     """Determine license from file at path.
 
     Args:
         path: filepath to determine the license
 
     Returns:
         dict: A dict with the following keys:
 
         - licenses ([str]): associated detected licenses to path
         - path (bytes): content filepath
 
     """
     try:
         properties = subprocess.check_output(['nomossa', path],
                                              universal_newlines=True)
         if properties:
             res = properties.rstrip().split(' contains license(s) ')
             licenses = res[1].split(',')
         else:
             licenses = []
 
         return {
             'licenses': licenses,
             'path': path,
         }
     except subprocess.CalledProcessError:
         if log:
             from os import path as __path
             log.exception('Problem during license detection for sha1 %s' %
                           __path.basename(path))
         return {
             'licenses': [],
             'path': path,
         }
 
 
 class MixinFossologyLicenseIndexer:
     """Mixin fossology license indexer.
 
     See :class:`FossologyLicenseIndexer` and
     :class:`FossologyLicenseRangeIndexer`
 
     """
     ADDITIONAL_CONFIG = {
         'workdir': ('str', '/tmp/swh/indexer.fossology.license'),
         'tools': ('dict', {
             'name': 'nomos',
             'version': '3.1.0rc2-31-ga2cbb8c',
             'configuration': {
                 'command_line': 'nomossa <filepath>',
             },
         }),
         'write_batch_size': ('int', 1000),
     }
 
-    CONFIG_BASE_FILENAME = 'indexer/fossology_license'
+    CONFIG_BASE_FILENAME = 'indexer/fossology_license'  # type: Optional[str]
 
     def prepare(self):
         super().prepare()
         self.working_directory = self.config['workdir']
 
     def index(self, id, data):
         """Index sha1s' content and store result.
 
         Args:
             id (bytes): content's identifier
             raw_content (bytes): associated raw content to content id
 
         Returns:
             dict: A dict, representing a content_license, with keys:
 
             - id (bytes): content's identifier (sha1)
             - license (bytes): license in bytes
             - path (bytes): path
             - indexer_configuration_id (int): tool used to compute the output
 
         """
         assert isinstance(id, bytes)
         with write_to_temp(
                 filename=hashutil.hash_to_hex(id),  # use the id as pathname
                 data=data,
                 working_directory=self.working_directory) as content_path:
             properties = compute_license(path=content_path, log=self.log)
             properties.update({
                 'id': id,
                 'indexer_configuration_id': self.tool['id'],
             })
         return properties
 
     def persist_index_computations(self, results, policy_update):
         """Persist the results in storage.
 
         Args:
             results ([dict]): list of content_license, dict with the
               following keys:
 
               - id (bytes): content's identifier (sha1)
               - license (bytes): license in bytes
               - path (bytes): path
 
             policy_update ([str]): either 'update-dups' or 'ignore-dups' to
               respectively update duplicates or ignore them
 
         """
         self.idx_storage.content_fossology_license_add(
             results, conflict_update=(policy_update == 'update-dups'))
 
 
 class FossologyLicenseIndexer(
         MixinFossologyLicenseIndexer, ContentIndexer):
     """Indexer in charge of:
 
     - filtering out content already indexed
     - reading content from objstorage per the content's id (sha1)
     - computing {license, encoding} from that content
     - store result in storage
 
     """
     def filter(self, ids):
         """Filter out known sha1s and return only missing ones.
 
         """
         yield from self.idx_storage.content_fossology_license_missing((
             {
                 'id': sha1,
                 'indexer_configuration_id': self.tool['id'],
             } for sha1 in ids
         ))
 
 
 class FossologyLicenseRangeIndexer(
         MixinFossologyLicenseIndexer, ContentRangeIndexer):
     """FossologyLicense Range Indexer working on range of content identifiers.
 
     - filters out the non textual content
     - (optionally) filters out content already indexed (cf
       :meth:`.indexed_contents_in_range`)
     - reads content from objstorage per the content's id (sha1)
     - computes {mimetype, encoding} from that content
     - stores result in storage
 
     """
     def indexed_contents_in_range(self, start, end):
         """Retrieve indexed content id within range [start, end].
 
         Args:
             start (bytes): Starting bound from range identifier
             end (bytes): End range identifier
 
         Returns:
             dict: a dict with keys:
 
             - **ids** [bytes]: iterable of content ids within the range.
             - **next** (Optional[bytes]): The next range of sha1 starts at
               this sha1 if any
 
         """
         return self.idx_storage.content_fossology_license_get_range(
                 start, end, self.tool['id'])
diff --git a/swh/indexer/indexer.py b/swh/indexer/indexer.py
index 867e9ed..d3b881f 100644
--- a/swh/indexer/indexer.py
+++ b/swh/indexer/indexer.py
@@ -1,620 +1,622 @@
 # Copyright (C) 2016-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import abc
 import os
 import logging
 import shutil
 import tempfile
 import datetime
+
 from copy import deepcopy
 from contextlib import contextmanager
+from typing import Any, Dict, Tuple
 
 from swh.scheduler import get_scheduler
 from swh.scheduler import CONFIG as SWH_CONFIG
 
 from swh.storage import get_storage
 from swh.core.config import SWHConfig
 from swh.objstorage import get_objstorage
 from swh.objstorage.exc import ObjNotFoundError
 from swh.indexer.storage import get_indexer_storage, INDEXER_CFG_KEY
 from swh.model import hashutil
 from swh.core import utils
 
 
 @contextmanager
 def write_to_temp(filename, data, working_directory):
     """Write the sha1's content in a temporary file.
 
     Args:
         filename (str): one of sha1's many filenames
         data (bytes): the sha1's content to write in temporary
           file
 
     Returns:
         The path to the temporary file created. That file is
         filled in with the raw content's data.
 
     """
     os.makedirs(working_directory, exist_ok=True)
     temp_dir = tempfile.mkdtemp(dir=working_directory)
     content_path = os.path.join(temp_dir, filename)
 
     with open(content_path, 'wb') as f:
         f.write(data)
 
     yield content_path
     shutil.rmtree(temp_dir)
 
 
 class BaseIndexer(SWHConfig, metaclass=abc.ABCMeta):
     """Base class for indexers to inherit from.
 
     The main entry point is the :func:`run` function which is in
     charge of triggering the computations on the batch dict/ids
     received.
 
     Indexers can:
 
     - filter out ids whose data has already been indexed.
     - retrieve ids data from storage or objstorage
     - index this data depending on the object and store the result in
       storage.
 
     To implement a new object type indexer, inherit from the
     BaseIndexer and implement indexing:
 
     :meth:`~BaseIndexer.run`:
       object_ids are different depending on object. For example: sha1 for
       content, sha1_git for revision, directory, release, and id for origin
 
     To implement a new concrete indexer, inherit from the object level
     classes: :class:`ContentIndexer`, :class:`RevisionIndexer`,
     :class:`OriginIndexer`.
 
     Then you need to implement the following functions:
 
     :meth:`~BaseIndexer.filter`:
       filter out data already indexed (in storage).
 
     :meth:`~BaseIndexer.index_object`:
       compute index on id with data (retrieved from the storage or the
       objstorage by the id key) and return the resulting index computation.
 
     :meth:`~BaseIndexer.persist_index_computations`:
       persist the results of multiple index computations in the storage.
 
     The new indexer implementation can also override the following functions:
 
     :meth:`~BaseIndexer.prepare`:
       Configuration preparation for the indexer.  When overriding, this must
       call the `super().prepare()` instruction.
 
     :meth:`~BaseIndexer.check`:
       Configuration check for the indexer.  When overriding, this must call the
       `super().check()` instruction.
 
     :meth:`~BaseIndexer.register_tools`:
       This should return a dict of the tool(s) to use when indexing or
       filtering.
 
     """
     CONFIG = 'indexer/base'
 
     DEFAULT_CONFIG = {
         INDEXER_CFG_KEY: ('dict', {
             'cls': 'remote',
             'args': {
                 'url': 'http://localhost:5007/'
             }
         }),
         'storage': ('dict', {
             'cls': 'remote',
             'args': {
                 'url': 'http://localhost:5002/',
             }
         }),
         'objstorage': ('dict', {
             'cls': 'remote',
             'args': {
                 'url': 'http://localhost:5003/',
             }
         })
     }
 
-    ADDITIONAL_CONFIG = {}
+    ADDITIONAL_CONFIG = {}  # type: Dict[str, Tuple[str, Any]]
 
     USE_TOOLS = True
 
     catch_exceptions = True
     """Prevents exceptions in `index()` from raising too high. Set to False
     in tests to properly catch all exceptions."""
 
     def __init__(self, config=None, **kw):
         """Prepare and check that the indexer is ready to run.
 
         """
         super().__init__()
         if config is not None:
             self.config = config
         elif SWH_CONFIG:
             self.config = SWH_CONFIG.copy()
         else:
             config_keys = ('base_filename', 'config_filename',
                            'additional_configs', 'global_config')
             config_args = {k: v for k, v in kw.items() if k in config_keys}
             if self.ADDITIONAL_CONFIG:
                 config_args.setdefault('additional_configs', []).append(
                     self.ADDITIONAL_CONFIG)
             self.config = self.parse_config_file(**config_args)
         self.prepare()
         self.check()
         self.log.debug('%s: config=%s', self, self.config)
 
     def prepare(self):
         """Prepare the indexer's needed runtime configuration.
            Without this step, the indexer cannot possibly run.
 
         """
         config_storage = self.config.get('storage')
         if config_storage:
             self.storage = get_storage(**config_storage)
 
         objstorage = self.config['objstorage']
         self.objstorage = get_objstorage(objstorage['cls'],
                                          objstorage['args'])
 
         idx_storage = self.config[INDEXER_CFG_KEY]
         self.idx_storage = get_indexer_storage(**idx_storage)
 
         _log = logging.getLogger('requests.packages.urllib3.connectionpool')
         _log.setLevel(logging.WARN)
         self.log = logging.getLogger('swh.indexer')
 
         if self.USE_TOOLS:
             self.tools = list(self.register_tools(
                 self.config.get('tools', [])))
         self.results = []
 
     @property
     def tool(self):
         return self.tools[0]
 
     def check(self):
         """Check the indexer's configuration is ok before proceeding.
            If ok, does nothing. If not raise error.
 
         """
         if self.USE_TOOLS and not self.tools:
             raise ValueError('Tools %s is unknown, cannot continue' %
                              self.tools)
 
     def _prepare_tool(self, tool):
         """Prepare the tool dict to be compliant with the storage api.
 
         """
         return {'tool_%s' % key: value for key, value in tool.items()}
 
     def register_tools(self, tools):
         """Permit to register tools to the storage.
 
            Add a sensible default which can be overridden if not
            sufficient.  (For now, all indexers use only one tool)
 
            Expects the self.config['tools'] property to be set with
            one or more tools.
 
         Args:
             tools (dict/[dict]): Either a dict or a list of dict.
 
         Returns:
             list: List of dicts with additional id key.
 
         Raises:
             ValueError: if not a list nor a dict.
 
         """
         if isinstance(tools, list):
             tools = list(map(self._prepare_tool, tools))
         elif isinstance(tools, dict):
             tools = [self._prepare_tool(tools)]
         else:
             raise ValueError('Configuration tool(s) must be a dict or list!')
 
         if tools:
             return self.idx_storage.indexer_configuration_add(tools)
         else:
             return []
 
     def index(self, id, data):
         """Index computation for the id and associated raw data.
 
         Args:
             id (bytes): identifier
             data (bytes): id's data from storage or objstorage depending on
               object type
 
         Returns:
             dict: a dict that makes sense for the
             :meth:`.persist_index_computations` method.
 
         """
         raise NotImplementedError()
 
     def filter(self, ids):
         """Filter missing ids for that particular indexer.
 
         Args:
             ids ([bytes]): list of ids
 
         Yields:
             iterator of missing ids
 
         """
         yield from ids
 
     @abc.abstractmethod
     def persist_index_computations(self, results, policy_update):
         """Persist the computation resulting from the index.
 
         Args:
 
             results ([result]): List of results. One result is the
               result of the index function.
             policy_update ([str]): either 'update-dups' or 'ignore-dups' to
               respectively update duplicates or ignore them
 
         Returns:
             None
 
         """
         pass
 
     def next_step(self, results, task):
         """Do something else with computations results (e.g. send to another
         queue, ...).
 
         (This is not an abstractmethod since it is optional).
 
         Args:
             results ([result]): List of results (dict) as returned
               by index function.
             task (dict): a dict in the form expected by
               `scheduler.backend.SchedulerBackend.create_tasks`
               without `next_run`, plus an optional `result_name` key.
 
         Returns:
             None
 
         """
         if task:
             if getattr(self, 'scheduler', None):
                 scheduler = self.scheduler
             else:
                 scheduler = get_scheduler(**self.config['scheduler'])
             task = deepcopy(task)
             result_name = task.pop('result_name', None)
             task['next_run'] = datetime.datetime.now()
             if result_name:
                 task['arguments']['kwargs'][result_name] = self.results
             scheduler.create_tasks([task])
 
     @abc.abstractmethod
     def run(self, ids, policy_update,
             next_step=None, **kwargs):
         """Given a list of ids:
 
         - retrieves the data from the storage
         - executes the indexing computations
         - stores the results (according to policy_update)
 
         Args:
             ids ([bytes]): id's identifier list
             policy_update (str): either 'update-dups' or 'ignore-dups' to
               respectively update duplicates or ignore them
             next_step (dict): a dict in the form expected by
               `scheduler.backend.SchedulerBackend.create_tasks`
               without `next_run`, plus a `result_name` key.
             **kwargs: passed to the `index` method
 
         """
         pass
 
 
 class ContentIndexer(BaseIndexer):
     """A content indexer working on a list of ids directly.
 
     To work on indexer range, use the :class:`ContentRangeIndexer`
     instead.
 
     Note: :class:`ContentIndexer` is not an instantiable object. To
     use it, one should inherit from this class and override the
     methods mentioned in the :class:`BaseIndexer` class.
 
     """
 
     def run(self, ids, policy_update,
             next_step=None, **kwargs):
         """Given a list of ids:
 
         - retrieve the content from the storage
         - execute the indexing computations
         - store the results (according to policy_update)
 
         Args:
             ids (Iterable[Union[bytes, str]]): sha1's identifier list
             policy_update (str): either 'update-dups' or 'ignore-dups' to
                                  respectively update duplicates or ignore
                                  them
             next_step (dict): a dict in the form expected by
                         `scheduler.backend.SchedulerBackend.create_tasks`
                         without `next_run`, plus an optional `result_name` key.
             **kwargs: passed to the `index` method
 
         """
         ids = [hashutil.hash_to_bytes(id_) if isinstance(id_, str) else id_
                for id_ in ids]
         results = []
         try:
             for sha1 in ids:
                 try:
                     raw_content = self.objstorage.get(sha1)
                 except ObjNotFoundError:
                     self.log.warning('Content %s not found in objstorage' %
                                      hashutil.hash_to_hex(sha1))
                     continue
                 res = self.index(sha1, raw_content, **kwargs)
                 if res:  # If no results, skip it
                     results.append(res)
 
             self.persist_index_computations(results, policy_update)
             self.results = results
             return self.next_step(results, task=next_step)
         except Exception:
             if not self.catch_exceptions:
                 raise
             self.log.exception(
                 'Problem when reading contents metadata.')
 
 
 class ContentRangeIndexer(BaseIndexer):
     """A content range indexer.
 
     This expects as input a range of ids to index.
 
     To work on a list of ids, use the :class:`ContentIndexer` instead.
 
     Note: :class:`ContentRangeIndexer` is not an instantiable
     object. To use it, one should inherit from this class and override
     the methods mentioned in the :class:`BaseIndexer` class.
 
     """
     @abc.abstractmethod
     def indexed_contents_in_range(self, start, end):
         """Retrieve indexed contents within range [start, end].
 
         Args:
             start (bytes): Starting bound from range identifier
             end (bytes): End range identifier
 
         Yields:
             bytes: Content identifier present in the range ``[start, end]``
 
         """
         pass
 
     def _list_contents_to_index(self, start, end, indexed):
         """Compute from storage the new contents to index in the range [start,
            end]. The already indexed contents are skipped.
 
         Args:
             start (bytes): Starting bound from range identifier
             end (bytes): End range identifier
             indexed (Set[bytes]): Set of content already indexed.
 
         Yields:
             bytes: Identifier of contents to index.
 
         """
         if not isinstance(start, bytes) or not isinstance(end, bytes):
             raise TypeError('identifiers must be bytes, not %r and %r.' %
                             (start, end))
         while start:
             result = self.storage.content_get_range(start, end)
             contents = result['contents']
             for c in contents:
                 _id = hashutil.hash_to_bytes(c['sha1'])
                 if _id in indexed:
                     continue
                 yield _id
             start = result['next']
 
     def _index_contents(self, start, end, indexed, **kwargs):
         """Index the contents from within range [start, end]
 
         Args:
             start (bytes): Starting bound from range identifier
             end (bytes): End range identifier
             indexed (Set[bytes]): Set of content already indexed.
 
         Yields:
             dict: Data indexed to persist using the indexer storage
 
         """
         for sha1 in self._list_contents_to_index(start, end, indexed):
             try:
                 raw_content = self.objstorage.get(sha1)
             except ObjNotFoundError:
                 self.log.warning('Content %s not found in objstorage' %
                                  hashutil.hash_to_hex(sha1))
                 continue
             res = self.index(sha1, raw_content, **kwargs)
             if res:
                 if not isinstance(res['id'], bytes):
                     raise TypeError(
                         '%r.index should return ids as bytes, not %r' %
                         (self.__class__.__name__, res['id']))
                 yield res
 
     def _index_with_skipping_already_done(self, start, end):
         """Index not already indexed contents in range [start, end].
 
         Args:
             start** (Union[bytes, str]): Starting range identifier
             end (Union[bytes, str]): Ending range identifier
 
         Yields:
             bytes: Content identifier present in the range
             ``[start, end]`` which are not already indexed.
 
         """
         while start:
             indexed_page = self.indexed_contents_in_range(start, end)
             contents = indexed_page['ids']
             _end = contents[-1] if contents else end
             yield from self._index_contents(
                     start, _end, contents)
             start = indexed_page['next']
 
     def run(self, start, end, skip_existing=True, **kwargs):
         """Given a range of content ids, compute the indexing computations on
            the contents within. Either the indexer is incremental
            (filter out existing computed data) or not (compute
            everything from scratch).
 
         Args:
             start (Union[bytes, str]): Starting range identifier
             end (Union[bytes, str]): Ending range identifier
             skip_existing (bool): Skip existing indexed data
               (default) or not
             **kwargs: passed to the `index` method
 
         Returns:
             bool: True if data was indexed, False otherwise.
 
         """
         with_indexed_data = False
         try:
             if isinstance(start, str):
                 start = hashutil.hash_to_bytes(start)
             if isinstance(end, str):
                 end = hashutil.hash_to_bytes(end)
 
             if skip_existing:
                 gen = self._index_with_skipping_already_done(start, end)
             else:
                 gen = self._index_contents(start, end, indexed=[])
 
             for results in utils.grouper(gen,
                                          n=self.config['write_batch_size']):
                 self.persist_index_computations(
                     results, policy_update='update-dups')
                 with_indexed_data = True
         except Exception:
             if not self.catch_exceptions:
                 raise
             self.log.exception(
                 'Problem when computing metadata.')
         finally:
             return with_indexed_data
 
 
 class OriginIndexer(BaseIndexer):
     """An object type indexer, inherits from the :class:`BaseIndexer` and
     implements Origin indexing using the run method
 
     Note: the :class:`OriginIndexer` is not an instantiable object.
     To use it in another context one should inherit from this class
     and override the methods mentioned in the :class:`BaseIndexer`
     class.
 
     """
     def run(self, origin_urls, policy_update='update-dups',
             next_step=None, **kwargs):
         """Given a list of origin ids:
 
         - retrieve origins from storage
         - execute the indexing computations
         - store the results (according to policy_update)
 
         Args:
             ids ([Union[int, Tuple[str, bytes]]]): list of origin ids or
               (type, url) tuples.
             policy_update (str): either 'update-dups' or 'ignore-dups' to
               respectively update duplicates (default) or ignore them
             next_step (dict): a dict in the form expected by
               `scheduler.backend.SchedulerBackend.create_tasks` without
               `next_run`, plus an optional `result_name` key.
             parse_ids (bool): Do we need to parse id or not (default)
             **kwargs: passed to the `index` method
 
         """
         results = self.index_list(origin_urls, **kwargs)
 
         self.persist_index_computations(results, policy_update)
         self.results = results
         return self.next_step(results, task=next_step)
 
     def index_list(self, origins, **kwargs):
         results = []
         for origin in origins:
             try:
                 res = self.index(origin, **kwargs)
                 if res:  # If no results, skip it
                     results.append(res)
             except Exception:
                 if not self.catch_exceptions:
                     raise
                 self.log.exception(
                     'Problem when processing origin %s',
                     origin['id'])
         return results
 
 
 class RevisionIndexer(BaseIndexer):
     """An object type indexer, inherits from the :class:`BaseIndexer` and
     implements Revision indexing using the run method
 
     Note: the :class:`RevisionIndexer` is not an instantiable object.
     To use it in another context one should inherit from this class
     and override the methods mentioned in the :class:`BaseIndexer`
     class.
 
     """
     def run(self, ids, policy_update, next_step=None):
         """Given a list of sha1_gits:
 
         - retrieve revisions from storage
         - execute the indexing computations
         - store the results (according to policy_update)
 
         Args:
             ids ([bytes or str]): sha1_git's identifier list
             policy_update (str): either 'update-dups' or 'ignore-dups' to
               respectively update duplicates or ignore them
 
         """
         results = []
         ids = [hashutil.hash_to_bytes(id_) if isinstance(id_, str) else id_
                for id_ in ids]
         revs = self.storage.revision_get(ids)
 
         for rev in revs:
             if not rev:
                 self.log.warning('Revisions %s not found in storage' %
                                  list(map(hashutil.hash_to_hex, ids)))
                 continue
             try:
                 res = self.index(rev)
                 if res:  # If no results, skip it
                     results.append(res)
             except Exception:
                 if not self.catch_exceptions:
                     raise
                 self.log.exception(
                         'Problem when processing revision')
         self.persist_index_computations(results, policy_update)
         self.results = results
         return self.next_step(results, task=next_step)
diff --git a/swh/indexer/metadata_dictionary/base.py b/swh/indexer/metadata_dictionary/base.py
index 9bc0ef5..4276dd2 100644
--- a/swh/indexer/metadata_dictionary/base.py
+++ b/swh/indexer/metadata_dictionary/base.py
@@ -1,211 +1,213 @@
 # Copyright (C) 2017-2019  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import abc
 import json
 import logging
 
+from typing import List
+
 from swh.indexer.codemeta import SCHEMA_URI
 from swh.indexer.codemeta import compact
 
 
 def merge_values(v1, v2):
     """If v1 and v2 are of the form `{"@list": l1}` and `{"@list": l2}`,
     returns `{"@list": l1 + l2}`.
     Otherwise, make them lists (if they are not already) and concatenate
     them.
 
     >>> merge_values('a', 'b')
     ['a', 'b']
     >>> merge_values(['a', 'b'], 'c')
     ['a', 'b', 'c']
     >>> merge_values({'@list': ['a', 'b']}, {'@list': ['c']})
     {'@list': ['a', 'b', 'c']}
     """
     if v1 is None:
         return v2
     elif v2 is None:
         return v1
     elif isinstance(v1, dict) and set(v1) == {'@list'}:
         assert isinstance(v1['@list'], list)
         if isinstance(v2, dict) and set(v2) == {'@list'}:
             assert isinstance(v2['@list'], list)
             return {'@list': v1['@list'] + v2['@list']}
         else:
             raise ValueError('Cannot merge %r and %r' % (v1, v2))
     else:
         if isinstance(v2, dict) and '@list' in v2:
             raise ValueError('Cannot merge %r and %r' % (v1, v2))
         if not isinstance(v1, list):
             v1 = [v1]
         if not isinstance(v2, list):
             v2 = [v2]
         return v1 + v2
 
 
 class BaseMapping(metaclass=abc.ABCMeta):
     """Base class for mappings to inherit from
 
     To implement a new mapping:
 
     - inherit this class
     - override translate function
     """
     def __init__(self, log_suffix=''):
         self.log_suffix = log_suffix
         self.log = logging.getLogger('%s.%s' % (
             self.__class__.__module__,
             self.__class__.__name__))
 
     @property
     @abc.abstractmethod
     def name(self):
         """A name of this mapping, used as an identifier in the
         indexer storage."""
         pass
 
     @classmethod
     @abc.abstractmethod
     def detect_metadata_files(cls, files):
         """
         Detects files potentially containing metadata
 
         Args:
             file_entries (list): list of files
 
         Returns:
             list: list of sha1 (possibly empty)
         """
         pass
 
     @abc.abstractmethod
     def translate(self, file_content):
         pass
 
     def normalize_translation(self, metadata):
         return compact(metadata)
 
 
 class SingleFileMapping(BaseMapping):
     """Base class for all mappings that use a single file as input."""
 
     @property
     @abc.abstractmethod
     def filename(self):
         """The .json file to extract metadata from."""
         pass
 
     @classmethod
     def detect_metadata_files(cls, file_entries):
         for entry in file_entries:
             if entry['name'] == cls.filename:
                 return [entry['sha1']]
         return []
 
 
 class DictMapping(BaseMapping):
     """Base class for mappings that take as input a file that is mostly
     a key-value store (eg. a shallow JSON dict)."""
 
-    string_fields = []
+    string_fields = []  # type: List[str]
     '''List of fields that are simple strings, and don't need any
     normalization.'''
 
     @property
     @abc.abstractmethod
     def mapping(self):
         """A translation dict to map dict keys into a canonical name."""
         pass
 
     @staticmethod
     def _normalize_method_name(name):
         return name.replace('-', '_')
 
     @classmethod
     def supported_terms(cls):
         return {
             term for (key, term) in cls.mapping.items()
             if key in cls.string_fields
             or hasattr(cls, 'translate_' + cls._normalize_method_name(key))
             or hasattr(cls, 'normalize_' + cls._normalize_method_name(key))}
 
     def _translate_dict(self, content_dict, *, normalize=True):
         """
         Translates content  by parsing content from a dict object
         and translating with the appropriate mapping
 
         Args:
             content_dict (dict): content dict to translate
 
         Returns:
             dict: translated metadata in json-friendly form needed for
             the indexer
 
         """
         translated_metadata = {'@type': SCHEMA_URI + 'SoftwareSourceCode'}
         for k, v in content_dict.items():
             # First, check if there is a specific translation
             # method for this key
             translation_method = getattr(
                 self, 'translate_' + self._normalize_method_name(k), None)
             if translation_method:
                 translation_method(translated_metadata, v)
             elif k in self.mapping:
                 # if there is no method, but the key is known from the
                 # crosswalk table
                 codemeta_key = self.mapping[k]
 
                 # if there is a normalization method, use it on the value
                 normalization_method = getattr(
                     self, 'normalize_' + self._normalize_method_name(k), None)
                 if normalization_method:
                     v = normalization_method(v)
                 elif k in self.string_fields and isinstance(v, str):
                     pass
                 elif k in self.string_fields and isinstance(v, list):
                     v = [x for x in v if isinstance(x, str)]
                 else:
                     continue
 
                 # set the translation metadata with the normalized value
                 if codemeta_key in translated_metadata:
                     translated_metadata[codemeta_key] = merge_values(
                         translated_metadata[codemeta_key], v)
                 else:
                     translated_metadata[codemeta_key] = v
         if normalize:
             return self.normalize_translation(translated_metadata)
         else:
             return translated_metadata
 
 
 class JsonMapping(DictMapping, SingleFileMapping):
     """Base class for all mappings that use a JSON file as input."""
 
     def translate(self, raw_content):
         """
         Translates content by parsing content from a bytestring containing
         json data and translating with the appropriate mapping
 
         Args:
             raw_content (bytes): raw content to translate
 
         Returns:
             dict: translated metadata in json-friendly form needed for
             the indexer
 
         """
         try:
             raw_content = raw_content.decode()
         except UnicodeDecodeError:
             self.log.warning('Error unidecoding from %s', self.log_suffix)
             return
         try:
             content_dict = json.loads(raw_content)
         except json.JSONDecodeError:
             self.log.warning('Error unjsoning from %s', self.log_suffix)
             return
         if isinstance(content_dict, dict):
             return self._translate_dict(content_dict)
diff --git a/swh/indexer/mimetype.py b/swh/indexer/mimetype.py
index afe700e..d8dda33 100644
--- a/swh/indexer/mimetype.py
+++ b/swh/indexer/mimetype.py
@@ -1,145 +1,147 @@
 # Copyright (C) 2016-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import magic
 
+from typing import Optional
+
 from .indexer import ContentIndexer, ContentRangeIndexer
 
 if not hasattr(magic.Magic, 'from_buffer'):
     raise ImportError(
         'Expected "import magic" to import python-magic, but file_magic '
         'was imported instead.')
 
 
 def compute_mimetype_encoding(raw_content):
     """Determine mimetype and encoding from the raw content.
 
     Args:
         raw_content (bytes): content's raw data
 
     Returns:
         dict: mimetype and encoding key and corresponding values
         (as bytes).
 
     """
     m = magic.Magic(mime=True, mime_encoding=True)
     res = m.from_buffer(raw_content)
     (mimetype, encoding) = res.split('; charset=')
     return {
         'mimetype': mimetype,
         'encoding': encoding,
     }
 
 
 class MixinMimetypeIndexer:
     """Mixin mimetype indexer.
 
     See :class:`MimetypeIndexer` and :class:`MimetypeRangeIndexer`
 
     """
     ADDITIONAL_CONFIG = {
         'tools': ('dict', {
             'name': 'file',
             'version': '1:5.30-1+deb9u1',
             'configuration': {
                 "type": "library",
                 "debian-package": "python3-magic"
             },
         }),
         'write_batch_size': ('int', 1000),
     }
 
-    CONFIG_BASE_FILENAME = 'indexer/mimetype'
+    CONFIG_BASE_FILENAME = 'indexer/mimetype'  # type: Optional[str]
 
     def index(self, id, data):
         """Index sha1s' content and store result.
 
         Args:
             id (bytes): content's identifier
             data (bytes): raw content in bytes
 
         Returns:
             dict: content's mimetype; dict keys being
 
             - **id** (bytes): content's identifier (sha1)
             - **mimetype** (bytes): mimetype in bytes
             - **encoding** (bytes): encoding in bytes
 
         """
         properties = compute_mimetype_encoding(data)
         properties.update({
             'id': id,
             'indexer_configuration_id': self.tool['id'],
             })
         return properties
 
     def persist_index_computations(self, results, policy_update):
         """Persist the results in storage.
 
         Args:
             results ([dict]): list of content's mimetype dicts
               (see :meth:`.index`)
 
             policy_update ([str]): either 'update-dups' or 'ignore-dups' to
                respectively update duplicates or ignore them
 
         """
         self.idx_storage.content_mimetype_add(
             results, conflict_update=(policy_update == 'update-dups'))
 
 
 class MimetypeIndexer(MixinMimetypeIndexer, ContentIndexer):
     """Mimetype Indexer working on list of content identifiers.
 
     It:
 
     - (optionally) filters out content already indexed (cf.
       :meth:`.filter`)
     - reads content from objstorage per the content's id (sha1)
     - computes {mimetype, encoding} from that content
     - stores result in storage
 
     """
     def filter(self, ids):
         """Filter out known sha1s and return only missing ones.
 
         """
         yield from self.idx_storage.content_mimetype_missing((
             {
                 'id': sha1,
                 'indexer_configuration_id': self.tool['id'],
             } for sha1 in ids
         ))
 
 
 class MimetypeRangeIndexer(MixinMimetypeIndexer, ContentRangeIndexer):
     """Mimetype Range Indexer working on range of content identifiers.
 
     It:
 
     - (optionally) filters out content already indexed (cf
       :meth:`.indexed_contents_in_range`)
     - reads content from objstorage per the content's id (sha1)
     - computes {mimetype, encoding} from that content
     - stores result in storage
 
     """
     def indexed_contents_in_range(self, start, end):
         """Retrieve indexed content id within range [start, end].
 
         Args:
             start (bytes): Starting bound from range identifier
             end (bytes): End range identifier
 
         Returns:
             dict: a dict with keys:
 
             - **ids** [bytes]: iterable of content ids within the range.
             - **next** (Optional[bytes]): The next range of sha1 starts at
               this sha1 if any
 
         """
         return self.idx_storage.content_mimetype_get_range(
             start, end, self.tool['id'])
diff --git a/swh/indexer/py.typed b/swh/indexer/py.typed
new file mode 100644
index 0000000..1242d43
--- /dev/null
+++ b/swh/indexer/py.typed
@@ -0,0 +1 @@
+# Marker file for PEP 561.
diff --git a/swh/indexer/tests/conftest.py b/swh/indexer/tests/conftest.py
index 5f651bf..78f1975 100644
--- a/swh/indexer/tests/conftest.py
+++ b/swh/indexer/tests/conftest.py
@@ -1,71 +1,71 @@
 from datetime import timedelta
 from unittest.mock import patch
 
 import pytest
 
 from swh.objstorage import get_objstorage
 from swh.scheduler.tests.conftest import *  # noqa
 from swh.storage.in_memory import Storage
 
 from swh.indexer.storage.in_memory import IndexerStorage
 
 from .utils import fill_storage, fill_obj_storage
 
 
 TASK_NAMES = ['revision_intrinsic_metadata', 'origin_intrinsic_metadata']
 
 
 @pytest.fixture
 def indexer_scheduler(swh_scheduler):
     for taskname in TASK_NAMES:
         swh_scheduler.create_task_type({
             'type': taskname,
             'description': 'The {} indexer testing task'.format(taskname),
             'backend_name': 'swh.indexer.tests.tasks.{}'.format(taskname),
             'default_interval': timedelta(days=1),
             'min_interval': timedelta(hours=6),
             'max_interval': timedelta(days=12),
             'num_retries': 3,
         })
     return swh_scheduler
 
 
 @pytest.fixture
 def idx_storage():
     """An instance of swh.indexer.storage.in_memory.IndexerStorage that
     gets injected into all indexers classes."""
     idx_storage = IndexerStorage()
     with patch('swh.indexer.storage.in_memory.IndexerStorage') \
             as idx_storage_mock:
         idx_storage_mock.return_value = idx_storage
         yield idx_storage
 
 
 @pytest.fixture
 def storage():
     """An instance of swh.storage.in_memory.Storage that gets injected
     into all indexers classes."""
     storage = Storage()
     fill_storage(storage)
     with patch('swh.storage.in_memory.Storage') as storage_mock:
         storage_mock.return_value = storage
         yield storage
 
 
 @pytest.fixture
 def obj_storage():
     """An instance of swh.objstorage.objstorage_in_memory.InMemoryObjStorage
     that gets injected into all indexers classes."""
     objstorage = get_objstorage('memory', {})
     fill_obj_storage(objstorage)
     with patch.dict('swh.objstorage._STORAGE_CLASSES',
                     {'memory': lambda: objstorage}):
         yield objstorage
 
 
-@pytest.fixture(scope='session')
+@pytest.fixture(scope='session')  # type: ignore  # expected redefinition
 def celery_includes():
     return [
         'swh.indexer.tests.tasks',
         'swh.indexer.tasks',
     ]
diff --git a/swh/indexer/tests/storage/test_storage.py b/swh/indexer/tests/storage/test_storage.py
index b11806a..9ffee2a 100644
--- a/swh/indexer/tests/storage/test_storage.py
+++ b/swh/indexer/tests/storage/test_storage.py
@@ -1,1994 +1,1994 @@
 # Copyright (C) 2015-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import os
 import threading
 import unittest
 
 import pytest
 from hypothesis import given
 
 from swh.model.hashutil import hash_to_bytes
 
 from swh.indexer.storage import get_indexer_storage, MAPPING_NAMES
 from swh.core.db.tests.db_testing import SingleDbTestFixture
 from swh.indexer.tests.storage.generate_data_test import (
     gen_content_mimetypes, gen_content_fossology_licenses
 )
 from swh.indexer.tests.storage import SQL_DIR
 from swh.indexer.metadata_dictionary import MAPPINGS
 
 TOOLS = [
     {
         'tool_name': 'universal-ctags',
         'tool_version': '~git7859817b',
         'tool_configuration': {
             "command_line": "ctags --fields=+lnz --sort=no --links=no "
                             "--output-format=json <filepath>"}
     },
     {
         'tool_name': 'swh-metadata-translator',
         'tool_version': '0.0.1',
         'tool_configuration': {"type": "local", "context": "NpmMapping"},
     },
     {
         'tool_name': 'swh-metadata-detector',
         'tool_version': '0.0.1',
         'tool_configuration': {
             "type": "local", "context": ["NpmMapping", "CodemetaMapping"]},
     },
     {
         'tool_name': 'swh-metadata-detector2',
         'tool_version': '0.0.1',
         'tool_configuration': {
             "type": "local", "context": ["NpmMapping", "CodemetaMapping"]},
     },
     {
         'tool_name': 'file',
         'tool_version': '5.22',
         'tool_configuration': {"command_line": "file --mime <filepath>"},
     },
     {
         'tool_name': 'pygments',
         'tool_version': '2.0.1+dfsg-1.1+deb8u1',
         'tool_configuration': {
             "type": "library", "debian-package": "python3-pygments"},
     },
     {
         'tool_name': 'pygments',
         'tool_version': '2.0.1+dfsg-1.1+deb8u1',
         'tool_configuration': {
             "type": "library",
             "debian-package": "python3-pygments",
             "max_content_size": 10240
         },
     },
     {
         'tool_name': 'nomos',
         'tool_version': '3.1.0rc2-31-ga2cbb8c',
         'tool_configuration': {"command_line": "nomossa <filepath>"},
     }
 ]
 
 
 @pytest.mark.db
 class BasePgTestStorage(SingleDbTestFixture):
     """Base test class for most indexer tests.
 
     It adds support for Storage testing to the SingleDbTestFixture class.
     It will also build the database from the swh-indexed/sql/*.sql files.
     """
 
     TEST_DB_NAME = 'softwareheritage-test-indexer'
     TEST_DB_DUMP = os.path.join(SQL_DIR, '*.sql')
 
     def setUp(self):
         super().setUp()
         self.storage_config = {
             'cls': 'local',
             'args': {
                 'db': 'dbname=%s' % self.TEST_DB_NAME,
             },
         }
 
     def tearDown(self):
         self.reset_storage_tables()
         self.storage = None
         super().tearDown()
 
     def reset_storage_tables(self):
         excluded = {'indexer_configuration'}
         self.reset_db_tables(self.TEST_DB_NAME, excluded=excluded)
 
         db = self.test_db[self.TEST_DB_NAME]
         db.conn.commit()
 
 
 def gen_generic_endpoint_tests(endpoint_type, tool_name,
                                example_data1, example_data2):
     def rename(f):
         f.__name__ = 'test_' + endpoint_type + f.__name__
         return f
 
     def endpoint(self, endpoint_name):
         return getattr(self.storage, endpoint_type + '_' + endpoint_name)
 
     @rename
     def missing(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         query = [
             {
                 'id': self.sha1_1,
                 'indexer_configuration_id': tool_id,
             },
             {
                 'id': self.sha1_2,
                 'indexer_configuration_id': tool_id,
             }]
 
         # when
         actual_missing = endpoint(self, 'missing')(query)
 
         # then
         self.assertEqual(list(actual_missing), [
             self.sha1_1,
             self.sha1_2,
         ])
 
         # given
         endpoint(self, 'add')([{
             'id': self.sha1_2,
             **example_data1,
             'indexer_configuration_id': tool_id,
         }])
 
         # when
         actual_missing = endpoint(self, 'missing')(query)
 
         # then
         self.assertEqual(list(actual_missing), [self.sha1_1])
 
     @rename
     def add__drop_duplicate(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         data_v1 = {
             'id': self.sha1_2,
             **example_data1,
             'indexer_configuration_id': tool_id,
         }
 
         # given
         endpoint(self, 'add')([data_v1])
 
         # when
         actual_data = list(endpoint(self, 'get')([self.sha1_2]))
 
         # then
         expected_data_v1 = [{
             'id': self.sha1_2,
             **example_data1,
             'tool': self.tools[tool_name],
         }]
         self.assertEqual(actual_data, expected_data_v1)
 
         # given
         data_v2 = data_v1.copy()
         data_v2.update(example_data2)
 
         endpoint(self, 'add')([data_v2])
 
         actual_data = list(endpoint(self, 'get')([self.sha1_2]))
 
         # data did not change as the v2 was dropped.
         self.assertEqual(actual_data, expected_data_v1)
 
     @rename
     def add__update_in_place_duplicate(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         data_v1 = {
             'id': self.sha1_2,
             **example_data1,
             'indexer_configuration_id': tool_id,
         }
 
         # given
         endpoint(self, 'add')([data_v1])
 
         # when
         actual_data = list(endpoint(self, 'get')([self.sha1_2]))
 
         expected_data_v1 = [{
             'id': self.sha1_2,
             **example_data1,
             'tool': self.tools[tool_name],
         }]
 
         # then
         self.assertEqual(actual_data, expected_data_v1)
 
         # given
         data_v2 = data_v1.copy()
         data_v2.update(example_data2)
 
         endpoint(self, 'add')([data_v2], conflict_update=True)
 
         actual_data = list(endpoint(self, 'get')([self.sha1_2]))
 
         expected_data_v2 = [{
             'id': self.sha1_2,
             **example_data2,
             'tool': self.tools[tool_name],
         }]
 
         # data did change as the v2 was used to overwrite v1
         self.assertEqual(actual_data, expected_data_v2)
 
     @rename
     def add__update_in_place_deadlock(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         hashes = [
             hash_to_bytes(
                 '34973274ccef6ab4dfaaf86599792fa9c3fe4{:03d}'.format(i))
             for i in range(1000)]
 
         data_v1 = [
             {
                 'id': hash_,
                 **example_data1,
                 'indexer_configuration_id': tool_id,
             }
             for hash_ in hashes
         ]
         data_v2 = [
             {
                 'id': hash_,
                 **example_data2,
                 'indexer_configuration_id': tool_id,
             }
             for hash_ in hashes
         ]
 
         # Remove one item from each, so that both queries have to succeed for
         # all items to be in the DB.
         data_v2a = data_v2[1:]
         data_v2b = list(reversed(data_v2[0:-1]))
 
         # given
         endpoint(self, 'add')(data_v1)
 
         # when
         actual_data = list(endpoint(self, 'get')(hashes))
 
         expected_data_v1 = [
             {
                 'id': hash_,
                 **example_data1,
                 'tool': self.tools[tool_name],
             }
             for hash_ in hashes
         ]
 
         # then
         self.assertEqual(actual_data, expected_data_v1)
 
         # given
         def f1():
             endpoint(self, 'add')(data_v2a, conflict_update=True)
 
         def f2():
             endpoint(self, 'add')(data_v2b, conflict_update=True)
 
         t1 = threading.Thread(target=f1)
         t2 = threading.Thread(target=f2)
         t2.start()
         t1.start()
 
         t1.join()
         t2.join()
 
         actual_data = list(endpoint(self, 'get')(hashes))
 
         expected_data_v2 = [
             {
                 'id': hash_,
                 **example_data2,
                 'tool': self.tools[tool_name],
             }
             for hash_ in hashes
         ]
 
         self.assertCountEqual(actual_data, expected_data_v2)
 
     def add__duplicate_twice(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         data_rev1 = {
             'id': self.revision_id_2,
             **example_data1,
             'indexer_configuration_id': tool_id
         }
 
         data_rev2 = {
             'id': self.revision_id_2,
             **example_data2,
             'indexer_configuration_id': tool_id
         }
 
         # when
         endpoint(self, 'add')([data_rev1])
 
         with self.assertRaises(ValueError):
             endpoint(self, 'add')(
                 [data_rev2, data_rev2],
                 conflict_update=True)
 
         # then
         actual_data = list(endpoint(self, 'get')(
             [self.revision_id_2, self.revision_id_1]))
 
         expected_data = [{
             'id': self.revision_id_2,
             **example_data1,
             'tool': self.tools[tool_name]
         }]
         self.assertEqual(actual_data, expected_data)
 
     @rename
     def get(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         query = [self.sha1_2, self.sha1_1]
 
         data1 = {
             'id': self.sha1_2,
             **example_data1,
             'indexer_configuration_id': tool_id,
         }
 
         # when
         endpoint(self, 'add')([data1])
 
         # then
         actual_data = list(endpoint(self, 'get')(query))
 
         # then
         expected_data = [{
             'id': self.sha1_2,
             **example_data1,
             'tool': self.tools[tool_name]
         }]
 
         self.assertEqual(actual_data, expected_data)
 
     @rename
     def delete(self):
         # given
         tool_id = self.tools[tool_name]['id']
 
         query = [self.sha1_2, self.sha1_1]
 
         data1 = {
             'id': self.sha1_2,
             **example_data1,
             'indexer_configuration_id': tool_id,
         }
 
         # when
         endpoint(self, 'add')([data1])
         endpoint(self, 'delete')([
             {
                 'id': self.sha1_2,
                 'indexer_configuration_id': tool_id,
             }
         ])
 
         # then
         actual_data = list(endpoint(self, 'get')(query))
 
         # then
         self.assertEqual(actual_data, [])
 
     @rename
     def delete_nonexisting(self):
         tool_id = self.tools[tool_name]['id']
         endpoint(self, 'delete')([
             {
                 'id': self.sha1_2,
                 'indexer_configuration_id': tool_id,
             }
         ])
 
     return (
         missing,
         add__drop_duplicate,
         add__update_in_place_duplicate,
         add__update_in_place_deadlock,
         add__duplicate_twice,
         get,
         delete,
         delete_nonexisting,
     )
 
 
 class CommonTestStorage:
     """Base class for Indexer Storage testing.
 
     """
-    def setUp(self):
+    def setUp(self, *args, **kwargs):
         super().setUp()
         self.storage = get_indexer_storage(**self.storage_config)
         tools = self.storage.indexer_configuration_add(TOOLS)
         self.tools = {}
         for tool in tools:
             tool_name = tool['tool_name']
             while tool_name in self.tools:
                 tool_name += '_'
             self.tools[tool_name] = {
                 'id': tool['id'],
                 'name': tool['tool_name'],
                 'version': tool['tool_version'],
                 'configuration': tool['tool_configuration'],
             }
 
         self.sha1_1 = hash_to_bytes('34973274ccef6ab4dfaaf86599792fa9c3fe4689')
         self.sha1_2 = hash_to_bytes('61c2b3a30496d329e21af70dd2d7e097046d07b7')
         self.revision_id_1 = hash_to_bytes(
             '7026b7c1a2af56521e951c01ed20f255fa054238')
         self.revision_id_2 = hash_to_bytes(
             '7026b7c1a2af56521e9587659012345678904321')
         self.revision_id_3 = hash_to_bytes(
             '7026b7c1a2af56521e9587659012345678904320')
         self.origin_id_1 = 44434341
         self.origin_id_2 = 44434342
         self.origin_id_3 = 54974445
 
     def test_check_config(self):
         self.assertTrue(self.storage.check_config(check_write=True))
         self.assertTrue(self.storage.check_config(check_write=False))
 
     # generate content_mimetype tests
     (
         test_content_mimetype_missing,
         test_content_mimetype_add__drop_duplicate,
         test_content_mimetype_add__update_in_place_duplicate,
         test_content_mimetype_add__update_in_place_deadlock,
         test_content_mimetype_add__duplicate_twice,
         test_content_mimetype_get,
         _,  # content_mimetype_detete,
         _,  # content_mimetype_detete_nonexisting,
     ) = gen_generic_endpoint_tests(
         endpoint_type='content_mimetype',
         tool_name='file',
         example_data1={
             'mimetype': 'text/plain',
             'encoding': 'utf-8',
         },
         example_data2={
             'mimetype': 'text/html',
             'encoding': 'us-ascii',
         },
     )
 
     # content_language tests
     (
         test_content_language_missing,
         test_content_language_add__drop_duplicate,
         test_content_language_add__update_in_place_duplicate,
         test_content_language_add__update_in_place_deadlock,
         test_content_language_add__duplicate_twice,
         test_content_language_get,
         _,  # test_content_language_delete,
         _,  # test_content_language_delete_nonexisting,
     ) = gen_generic_endpoint_tests(
         endpoint_type='content_language',
         tool_name='pygments',
         example_data1={
             'lang': 'haskell',
         },
         example_data2={
             'lang': 'common-lisp',
         },
     )
 
     # content_ctags tests
     (
         test_content_ctags_missing,
         # the following tests are disabled because CTAGS behave differently
         _,  # test_content_ctags_add__drop_duplicate,
         _,  # test_content_ctags_add__update_in_place_duplicate,
         _,  # test_content_ctags_add__update_in_place_deadlock,
         _,  # test_content_ctags_add__duplicate_twice,
         _,  # test_content_ctags_get,
         _,  # test_content_ctags_delete,
         _,  # test_content_ctags_delete_nonexisting,
     ) = gen_generic_endpoint_tests(
         endpoint_type='content_ctags',
         tool_name='universal-ctags',
         example_data1={
             'ctags': [{
                 'name': 'done',
                 'kind': 'variable',
                 'line': 119,
                 'lang': 'OCaml',
             }]
         },
         example_data2={
             'ctags': [
                 {
                     'name': 'done',
                     'kind': 'variable',
                     'line': 100,
                     'lang': 'Python',
                 },
                 {
                     'name': 'main',
                     'kind': 'function',
                     'line': 119,
                     'lang': 'Python',
                 }]
         },
     )
 
     def test_content_ctags_search(self):
         # 1. given
         tool = self.tools['universal-ctags']
         tool_id = tool['id']
 
         ctag1 = {
             'id': self.sha1_1,
             'indexer_configuration_id': tool_id,
             'ctags': [
                 {
                     'name': 'hello',
                     'kind': 'function',
                     'line': 133,
                     'lang': 'Python',
                 },
                 {
                     'name': 'counter',
                     'kind': 'variable',
                     'line': 119,
                     'lang': 'Python',
                 },
                 {
                     'name': 'hello',
                     'kind': 'variable',
                     'line': 210,
                     'lang': 'Python',
                 },
             ]
         }
 
         ctag2 = {
             'id': self.sha1_2,
             'indexer_configuration_id': tool_id,
             'ctags': [
                 {
                     'name': 'hello',
                     'kind': 'variable',
                     'line': 100,
                     'lang': 'C',
                 },
                 {
                     'name': 'result',
                     'kind': 'variable',
                     'line': 120,
                     'lang': 'C',
                 },
             ]
         }
 
         self.storage.content_ctags_add([ctag1, ctag2])
 
         # 1. when
         actual_ctags = list(self.storage.content_ctags_search('hello',
                                                               limit=1))
 
         # 1. then
         self.assertEqual(actual_ctags, [
             {
                 'id': ctag1['id'],
                 'tool': tool,
                 'name': 'hello',
                 'kind': 'function',
                 'line': 133,
                 'lang': 'Python',
             }
         ])
 
         # 2. when
         actual_ctags = list(self.storage.content_ctags_search(
             'hello',
             limit=1,
             last_sha1=ctag1['id']))
 
         # 2. then
         self.assertEqual(actual_ctags, [
             {
                 'id': ctag2['id'],
                 'tool': tool,
                 'name': 'hello',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'C',
             }
         ])
 
         # 3. when
         actual_ctags = list(self.storage.content_ctags_search('hello'))
 
         # 3. then
         self.assertEqual(actual_ctags, [
             {
                 'id': ctag1['id'],
                 'tool': tool,
                 'name': 'hello',
                 'kind': 'function',
                 'line': 133,
                 'lang': 'Python',
             },
             {
                 'id': ctag1['id'],
                 'tool': tool,
                 'name': 'hello',
                 'kind': 'variable',
                 'line': 210,
                 'lang': 'Python',
             },
             {
                 'id': ctag2['id'],
                 'tool': tool,
                 'name': 'hello',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'C',
             },
         ])
 
         # 4. when
         actual_ctags = list(self.storage.content_ctags_search('counter'))
 
         # then
         self.assertEqual(actual_ctags, [{
             'id': ctag1['id'],
             'tool': tool,
             'name': 'counter',
             'kind': 'variable',
             'line': 119,
             'lang': 'Python',
         }])
 
         # 5. when
         actual_ctags = list(self.storage.content_ctags_search('result',
                                                               limit=1))
 
         # then
         self.assertEqual(actual_ctags, [{
             'id': ctag2['id'],
             'tool': tool,
             'name': 'result',
             'kind': 'variable',
             'line': 120,
             'lang': 'C',
         }])
 
     def test_content_ctags_search_no_result(self):
         actual_ctags = list(self.storage.content_ctags_search('counter'))
 
         self.assertEqual(actual_ctags, [])
 
     def test_content_ctags_add__add_new_ctags_added(self):
         # given
         tool = self.tools['universal-ctags']
         tool_id = tool['id']
 
         ctag_v1 = {
             'id': self.sha1_2,
             'indexer_configuration_id': tool_id,
             'ctags': [{
                 'name': 'done',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'Scheme',
             }]
         }
 
         # given
         self.storage.content_ctags_add([ctag_v1])
         self.storage.content_ctags_add([ctag_v1])  # conflict does nothing
 
         # when
         actual_ctags = list(self.storage.content_ctags_get(
             [self.sha1_2]))
 
         # then
         expected_ctags = [{
             'id': self.sha1_2,
             'name': 'done',
             'kind': 'variable',
             'line': 100,
             'lang': 'Scheme',
             'tool': tool,
         }]
 
         self.assertEqual(actual_ctags, expected_ctags)
 
         # given
         ctag_v2 = ctag_v1.copy()
         ctag_v2.update({
             'ctags': [
                 {
                     'name': 'defn',
                     'kind': 'function',
                     'line': 120,
                     'lang': 'Scheme',
                 }
             ]
         })
 
         self.storage.content_ctags_add([ctag_v2])
 
         expected_ctags = [
             {
                 'id': self.sha1_2,
                 'name': 'done',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'Scheme',
                 'tool': tool,
             }, {
                 'id': self.sha1_2,
                 'name': 'defn',
                 'kind': 'function',
                 'line': 120,
                 'lang': 'Scheme',
                 'tool': tool,
             }
         ]
 
         actual_ctags = list(self.storage.content_ctags_get(
             [self.sha1_2]))
 
         self.assertEqual(actual_ctags, expected_ctags)
 
     def test_content_ctags_add__update_in_place(self):
         # given
         tool = self.tools['universal-ctags']
         tool_id = tool['id']
 
         ctag_v1 = {
             'id': self.sha1_2,
             'indexer_configuration_id': tool_id,
             'ctags': [{
                 'name': 'done',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'Scheme',
             }]
         }
 
         # given
         self.storage.content_ctags_add([ctag_v1])
 
         # when
         actual_ctags = list(self.storage.content_ctags_get(
             [self.sha1_2]))
 
         # then
         expected_ctags = [
             {
                 'id': self.sha1_2,
                 'name': 'done',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'Scheme',
                 'tool': tool
             }
         ]
         self.assertEqual(actual_ctags, expected_ctags)
 
         # given
         ctag_v2 = ctag_v1.copy()
         ctag_v2.update({
             'ctags': [
                 {
                     'name': 'done',
                     'kind': 'variable',
                     'line': 100,
                     'lang': 'Scheme',
                 },
                 {
                     'name': 'defn',
                     'kind': 'function',
                     'line': 120,
                     'lang': 'Scheme',
                 }
             ]
         })
 
         self.storage.content_ctags_add([ctag_v2], conflict_update=True)
 
         actual_ctags = list(self.storage.content_ctags_get(
             [self.sha1_2]))
 
         # ctag did change as the v2 was used to overwrite v1
         expected_ctags = [
             {
                 'id': self.sha1_2,
                 'name': 'done',
                 'kind': 'variable',
                 'line': 100,
                 'lang': 'Scheme',
                 'tool': tool,
             },
             {
                 'id': self.sha1_2,
                 'name': 'defn',
                 'kind': 'function',
                 'line': 120,
                 'lang': 'Scheme',
                 'tool': tool,
             }
         ]
         self.assertEqual(actual_ctags, expected_ctags)
 
     # content_fossology_license tests
     (
         _,  # The endpoint content_fossology_license_missing does not exist
         # the following tests are disabled because fossology_license tests
         # behave differently
         _,  # test_content_fossology_license_add__drop_duplicate,
         _,  # test_content_fossology_license_add__update_in_place_duplicate,
         _,  # test_content_fossology_license_add__update_in_place_deadlock,
         _,  # test_content_metadata_add__duplicate_twice,
         _,  # test_content_fossology_license_get,
         _,  # test_content_fossology_license_delete,
         _,  # test_content_fossology_license_delete_nonexisting,
     ) = gen_generic_endpoint_tests(
         endpoint_type='content_fossology_license',
         tool_name='nomos',
         example_data1={
             'licenses': ['Apache-2.0'],
         },
         example_data2={
             'licenses': ['BSD-2-Clause'],
         },
     )
 
     def test_content_fossology_license_add__new_license_added(self):
         # given
         tool = self.tools['nomos']
         tool_id = tool['id']
 
         license_v1 = {
             'id': self.sha1_1,
             'licenses': ['Apache-2.0'],
             'indexer_configuration_id': tool_id,
         }
 
         # given
         self.storage.content_fossology_license_add([license_v1])
         # conflict does nothing
         self.storage.content_fossology_license_add([license_v1])
 
         # when
         actual_licenses = list(self.storage.content_fossology_license_get(
             [self.sha1_1]))
 
         # then
         expected_license = {
             self.sha1_1: [{
                 'licenses': ['Apache-2.0'],
                 'tool': tool,
             }]
         }
         self.assertEqual(actual_licenses, [expected_license])
 
         # given
         license_v2 = license_v1.copy()
         license_v2.update({
             'licenses': ['BSD-2-Clause'],
         })
 
         self.storage.content_fossology_license_add([license_v2])
 
         actual_licenses = list(self.storage.content_fossology_license_get(
             [self.sha1_1]))
 
         expected_license = {
             self.sha1_1: [{
                 'licenses': ['Apache-2.0', 'BSD-2-Clause'],
                 'tool': tool
             }]
         }
 
         # license did not change as the v2 was dropped.
         self.assertEqual(actual_licenses, [expected_license])
 
     # content_metadata tests
     (
         test_content_metadata_missing,
         test_content_metadata_add__drop_duplicate,
         test_content_metadata_add__update_in_place_duplicate,
         test_content_metadata_add__update_in_place_deadlock,
         test_content_metadata_add__duplicate_twice,
         test_content_metadata_get,
         _,  # test_content_metadata_delete,
         _,  # test_content_metadata_delete_nonexisting,
     ) = gen_generic_endpoint_tests(
         endpoint_type='content_metadata',
         tool_name='swh-metadata-detector',
         example_data1={
             'metadata': {
                 'other': {},
                 'codeRepository': {
                     'type': 'git',
                     'url': 'https://github.com/moranegg/metadata_test'
                 },
                 'description': 'Simple package.json test for indexer',
                 'name': 'test_metadata',
                 'version': '0.0.1'
             },
         },
         example_data2={
             'metadata': {
                 'other': {},
                 'name': 'test_metadata',
                 'version': '0.0.1'
             },
         },
     )
 
     # revision_intrinsic_metadata tests
     (
         test_revision_intrinsic_metadata_missing,
         test_revision_intrinsic_metadata_add__drop_duplicate,
         test_revision_intrinsic_metadata_add__update_in_place_duplicate,
         test_revision_intrinsic_metadata_add__update_in_place_deadlock,
         test_revision_intrinsic_metadata_add__duplicate_twice,
         test_revision_intrinsic_metadata_get,
         test_revision_intrinsic_metadata_delete,
         test_revision_intrinsic_metadata_delete_nonexisting,
     ) = gen_generic_endpoint_tests(
         endpoint_type='revision_intrinsic_metadata',
         tool_name='swh-metadata-detector',
         example_data1={
             'metadata': {
                 'other': {},
                 'codeRepository': {
                     'type': 'git',
                     'url': 'https://github.com/moranegg/metadata_test'
                 },
                 'description': 'Simple package.json test for indexer',
                 'name': 'test_metadata',
                 'version': '0.0.1'
             },
             'mappings': ['mapping1'],
         },
         example_data2={
             'metadata': {
                 'other': {},
                 'name': 'test_metadata',
                 'version': '0.0.1'
             },
             'mappings': ['mapping2'],
         },
     )
 
     def test_origin_intrinsic_metadata_get(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         metadata = {
             'version': None,
             'name': None,
         }
         metadata_rev = {
             'id': self.revision_id_2,
             'metadata': metadata,
             'mappings': ['mapping1'],
             'indexer_configuration_id': tool_id,
         }
         metadata_origin = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata,
             'indexer_configuration_id': tool_id,
             'mappings': ['mapping1'],
             'from_revision': self.revision_id_2,
             }
 
         # when
         self.storage.revision_intrinsic_metadata_add([metadata_rev])
         self.storage.origin_intrinsic_metadata_add([metadata_origin])
 
         # then
         actual_metadata = list(self.storage.origin_intrinsic_metadata_get(
             [self.origin_id_1, 42]))
 
         expected_metadata = [{
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata,
             'tool': self.tools['swh-metadata-detector'],
             'from_revision': self.revision_id_2,
             'mappings': ['mapping1'],
         }]
 
         self.assertEqual(actual_metadata, expected_metadata)
 
     def test_origin_intrinsic_metadata_delete(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         metadata = {
             'version': None,
             'name': None,
         }
         metadata_rev = {
             'id': self.revision_id_2,
             'metadata': metadata,
             'mappings': ['mapping1'],
             'indexer_configuration_id': tool_id,
         }
         metadata_origin = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata,
             'indexer_configuration_id': tool_id,
             'mappings': ['mapping1'],
             'from_revision': self.revision_id_2,
             }
         metadata_origin2 = metadata_origin.copy()
         metadata_origin2['id'] = self.origin_id_2
 
         # when
         self.storage.revision_intrinsic_metadata_add([metadata_rev])
         self.storage.origin_intrinsic_metadata_add([
             metadata_origin, metadata_origin2])
 
         self.storage.origin_intrinsic_metadata_delete([
             {
                 'id': self.origin_id_1,
                 'indexer_configuration_id': tool_id
             }
         ])
 
         # then
         actual_metadata = list(self.storage.origin_intrinsic_metadata_get(
             [self.origin_id_1, self.origin_id_2, 42]))
         for item in actual_metadata:
             item['indexer_configuration_id'] = item.pop('tool')['id']
         self.assertEqual(actual_metadata, [metadata_origin2])
 
     def test_origin_intrinsic_metadata_delete_nonexisting(self):
         tool_id = self.tools['swh-metadata-detector']['id']
         self.storage.origin_intrinsic_metadata_delete([
             {
                 'id': self.origin_id_1,
                 'indexer_configuration_id': tool_id
             }
         ])
 
     def test_origin_intrinsic_metadata_add_drop_duplicate(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         metadata_v1 = {
             'version': None,
             'name': None,
         }
         metadata_rev_v1 = {
             'id': self.revision_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata_v1.copy(),
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
         metadata_origin_v1 = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata_v1.copy(),
             'indexer_configuration_id': tool_id,
             'mappings': [],
             'from_revision': self.revision_id_1,
         }
 
         # given
         self.storage.revision_intrinsic_metadata_add([metadata_rev_v1])
         self.storage.origin_intrinsic_metadata_add([metadata_origin_v1])
 
         # when
         actual_metadata = list(self.storage.origin_intrinsic_metadata_get(
             [self.origin_id_1, 42]))
 
         expected_metadata_v1 = [{
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata_v1,
             'tool': self.tools['swh-metadata-detector'],
             'from_revision': self.revision_id_1,
             'mappings': [],
         }]
 
         self.assertEqual(actual_metadata, expected_metadata_v1)
 
         # given
         metadata_v2 = metadata_v1.copy()
         metadata_v2.update({
             'name': 'test_metadata',
             'author': 'MG',
         })
         metadata_rev_v2 = metadata_rev_v1.copy()
         metadata_origin_v2 = metadata_origin_v1.copy()
         metadata_rev_v2['metadata'] = metadata_v2
         metadata_origin_v2['metadata'] = metadata_v2
 
         self.storage.revision_intrinsic_metadata_add([metadata_rev_v2])
         self.storage.origin_intrinsic_metadata_add([metadata_origin_v2])
 
         # then
         actual_metadata = list(self.storage.origin_intrinsic_metadata_get(
             [self.origin_id_1]))
 
         # metadata did not change as the v2 was dropped.
         self.assertEqual(actual_metadata, expected_metadata_v1)
 
     def test_origin_intrinsic_metadata_add_update_in_place_duplicate(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         metadata_v1 = {
             'version': None,
             'name': None,
         }
         metadata_rev_v1 = {
             'id': self.revision_id_2,
             'metadata': metadata_v1,
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
         metadata_origin_v1 = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata_v1.copy(),
             'indexer_configuration_id': tool_id,
             'mappings': [],
             'from_revision': self.revision_id_2,
         }
 
         # given
         self.storage.revision_intrinsic_metadata_add([metadata_rev_v1])
         self.storage.origin_intrinsic_metadata_add([metadata_origin_v1])
 
         # when
         actual_metadata = list(self.storage.origin_intrinsic_metadata_get(
             [self.origin_id_1]))
 
         # then
         expected_metadata_v1 = [{
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata_v1,
             'tool': self.tools['swh-metadata-detector'],
             'from_revision': self.revision_id_2,
             'mappings': [],
         }]
         self.assertEqual(actual_metadata, expected_metadata_v1)
 
         # given
         metadata_v2 = metadata_v1.copy()
         metadata_v2.update({
             'name': 'test_update_duplicated_metadata',
             'author': 'MG',
         })
         metadata_rev_v2 = metadata_rev_v1.copy()
         metadata_origin_v2 = metadata_origin_v1.copy()
         metadata_rev_v2['metadata'] = metadata_v2
         metadata_origin_v2 = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/null',
             'metadata': metadata_v2.copy(),
             'indexer_configuration_id': tool_id,
             'mappings': ['npm'],
             'from_revision': self.revision_id_1,
         }
 
         self.storage.revision_intrinsic_metadata_add(
                 [metadata_rev_v2], conflict_update=True)
         self.storage.origin_intrinsic_metadata_add(
                 [metadata_origin_v2], conflict_update=True)
 
         actual_metadata = list(self.storage.origin_intrinsic_metadata_get(
             [self.origin_id_1]))
 
         expected_metadata_v2 = [{
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/null',
             'metadata': metadata_v2,
             'tool': self.tools['swh-metadata-detector'],
             'from_revision': self.revision_id_1,
             'mappings': ['npm'],
         }]
 
         # metadata did change as the v2 was used to overwrite v1
         self.assertEqual(actual_metadata, expected_metadata_v2)
 
     def test_origin_intrinsic_metadata_add__update_in_place_deadlock(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         ids = list(range(10))
 
         example_data1 = {
             'metadata': {
                 'version': None,
                 'name': None,
             },
             'mappings': [],
         }
         example_data2 = {
             'metadata': {
                 'version': 'v1.1.1',
                 'name': 'foo',
             },
             'mappings': [],
         }
 
         metadata_rev_v1 = {
             'id': self.revision_id_2,
             'metadata': {
                 'version': None,
                 'name': None,
             },
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
 
         data_v1 = [
             {
                 'id': id_,
                 'origin_url': 'file:///tmp/origin%d' % id_,
                 'from_revision': self.revision_id_2,
                 **example_data1,
                 'indexer_configuration_id': tool_id,
             }
             for id_ in ids
         ]
         data_v2 = [
             {
                 'id': id_,
                 'origin_url': 'file:///tmp/origin%d' % id_,
                 'from_revision': self.revision_id_2,
                 **example_data2,
                 'indexer_configuration_id': tool_id,
             }
             for id_ in ids
         ]
 
         # Remove one item from each, so that both queries have to succeed for
         # all items to be in the DB.
         data_v2a = data_v2[1:]
         data_v2b = list(reversed(data_v2[0:-1]))
 
         # given
         self.storage.revision_intrinsic_metadata_add([metadata_rev_v1])
         self.storage.origin_intrinsic_metadata_add(data_v1)
 
         # when
         actual_data = list(self.storage.origin_intrinsic_metadata_get(ids))
 
         expected_data_v1 = [
             {
                 'id': id_,
                 'origin_url': 'file:///tmp/origin%d' % id_,
                 'from_revision': self.revision_id_2,
                 **example_data1,
                 'tool': self.tools['swh-metadata-detector'],
             }
             for id_ in ids
         ]
 
         # then
         self.assertEqual(actual_data, expected_data_v1)
 
         # given
         def f1():
             self.storage.origin_intrinsic_metadata_add(
                 data_v2a, conflict_update=True)
 
         def f2():
             self.storage.origin_intrinsic_metadata_add(
                 data_v2b, conflict_update=True)
 
         t1 = threading.Thread(target=f1)
         t2 = threading.Thread(target=f2)
         t2.start()
         t1.start()
 
         t1.join()
         t2.join()
 
         actual_data = list(self.storage.origin_intrinsic_metadata_get(ids))
 
         expected_data_v2 = [
             {
                 'id': id_,
                 'origin_url': 'file:///tmp/origin%d' % id_,
                 'from_revision': self.revision_id_2,
                 **example_data2,
                 'tool': self.tools['swh-metadata-detector'],
             }
             for id_ in ids
         ]
 
         self.maxDiff = None
         self.assertCountEqual(actual_data, expected_data_v2)
 
     def test_origin_intrinsic_metadata_add__duplicate_twice(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         metadata = {
             'developmentStatus': None,
             'name': None,
         }
         metadata_rev = {
             'id': self.revision_id_2,
             'metadata': metadata,
             'mappings': ['mapping1'],
             'indexer_configuration_id': tool_id,
         }
         metadata_origin = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata,
             'indexer_configuration_id': tool_id,
             'mappings': ['mapping1'],
             'from_revision': self.revision_id_2,
             }
 
         # when
         self.storage.revision_intrinsic_metadata_add([metadata_rev])
 
         with self.assertRaises(ValueError):
             self.storage.origin_intrinsic_metadata_add([
                 metadata_origin, metadata_origin])
 
     def test_origin_intrinsic_metadata_search_fulltext(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         metadata1 = {
             'author': 'John Doe',
         }
         metadata1_rev = {
             'id': self.revision_id_1,
             'metadata': metadata1,
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
         metadata1_origin = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata1,
             'mappings': [],
             'indexer_configuration_id': tool_id,
             'from_revision': self.revision_id_1,
         }
         metadata2 = {
             'author': 'Jane Doe',
         }
         metadata2_rev = {
             'id': self.revision_id_2,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata2,
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
         metadata2_origin = {
             'id': self.origin_id_2,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata2,
             'mappings': [],
             'indexer_configuration_id': tool_id,
             'from_revision': self.revision_id_2,
         }
 
         # when
         self.storage.revision_intrinsic_metadata_add([metadata1_rev])
         self.storage.origin_intrinsic_metadata_add([metadata1_origin])
         self.storage.revision_intrinsic_metadata_add([metadata2_rev])
         self.storage.origin_intrinsic_metadata_add([metadata2_origin])
 
         # then
         search = self.storage.origin_intrinsic_metadata_search_fulltext
         self.assertCountEqual(
                 [res['id'] for res in search(['Doe'])],
                 [self.origin_id_1, self.origin_id_2])
         self.assertEqual(
                 [res['id'] for res in search(['John', 'Doe'])],
                 [self.origin_id_1])
         self.assertEqual(
                 [res['id'] for res in search(['John'])],
                 [self.origin_id_1])
         self.assertEqual(
                 [res['id'] for res in search(['John', 'Jane'])],
                 [])
 
     def test_origin_intrinsic_metadata_search_fulltext_rank(self):
         # given
         tool_id = self.tools['swh-metadata-detector']['id']
 
         # The following authors have "Random Person" to add some more content
         # to the JSON data, to work around normalization quirks when there
         # are few words (rank/(1+ln(nb_words)) is very sensitive to nb_words
         # for small values of nb_words).
         metadata1 = {
             'author': [
                 'Random Person',
                 'John Doe',
                 'Jane Doe',
             ]
         }
         metadata1_rev = {
             'id': self.revision_id_1,
             'metadata': metadata1,
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
         metadata1_origin = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata1,
             'mappings': [],
             'indexer_configuration_id': tool_id,
             'from_revision': self.revision_id_1,
         }
         metadata2 = {
             'author': [
                 'Random Person',
                 'Jane Doe',
             ]
         }
         metadata2_rev = {
             'id': self.revision_id_2,
             'metadata': metadata2,
             'mappings': [],
             'indexer_configuration_id': tool_id,
         }
         metadata2_origin = {
             'id': self.origin_id_2,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata2,
             'mappings': [],
             'indexer_configuration_id': tool_id,
             'from_revision': self.revision_id_2,
         }
 
         # when
         self.storage.revision_intrinsic_metadata_add([metadata1_rev])
         self.storage.origin_intrinsic_metadata_add([metadata1_origin])
         self.storage.revision_intrinsic_metadata_add([metadata2_rev])
         self.storage.origin_intrinsic_metadata_add([metadata2_origin])
 
         # then
         search = self.storage.origin_intrinsic_metadata_search_fulltext
         self.assertEqual(
                 [res['id'] for res in search(['Doe'])],
                 [self.origin_id_1, self.origin_id_2])
         self.assertEqual(
                 [res['id'] for res in search(['Doe'], limit=1)],
                 [self.origin_id_1])
         self.assertEqual(
                 [res['id'] for res in search(['John'])],
                 [self.origin_id_1])
         self.assertEqual(
                 [res['id'] for res in search(['Jane'])],
                 [self.origin_id_2, self.origin_id_1])
         self.assertEqual(
                 [res['id'] for res in search(['John', 'Jane'])],
                 [self.origin_id_1])
 
     def _fill_origin_intrinsic_metadata(self):
         tool1_id = self.tools['swh-metadata-detector']['id']
         tool2_id = self.tools['swh-metadata-detector2']['id']
 
         metadata1 = {
             '@context': 'foo',
             'author': 'John Doe',
         }
         metadata1_rev = {
             'id': self.revision_id_1,
             'metadata': metadata1,
             'mappings': ['npm'],
             'indexer_configuration_id': tool1_id,
         }
         metadata1_origin = {
             'id': self.origin_id_1,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata1,
             'mappings': ['npm'],
             'indexer_configuration_id': tool1_id,
             'from_revision': self.revision_id_1,
         }
         metadata2 = {
             '@context': 'foo',
             'author': 'Jane Doe',
         }
         metadata2_rev = {
             'id': self.revision_id_2,
             'metadata': metadata2,
             'mappings': ['npm', 'gemspec'],
             'indexer_configuration_id': tool2_id,
         }
         metadata2_origin = {
             'id': self.origin_id_2,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata2,
             'mappings': ['npm', 'gemspec'],
             'indexer_configuration_id': tool2_id,
             'from_revision': self.revision_id_2,
         }
         metadata3 = {
             '@context': 'foo',
         }
         metadata3_rev = {
             'id': self.revision_id_3,
             'metadata': metadata3,
             'mappings': ['npm', 'gemspec'],
             'indexer_configuration_id': tool2_id,
         }
         metadata3_origin = {
             'id': self.origin_id_3,
             'origin_url': 'file:///dev/zero',
             'metadata': metadata3,
             'mappings': ['pkg-info'],
             'indexer_configuration_id': tool2_id,
             'from_revision': self.revision_id_3,
         }
 
         self.storage.revision_intrinsic_metadata_add([metadata1_rev])
         self.storage.origin_intrinsic_metadata_add([metadata1_origin])
         self.storage.revision_intrinsic_metadata_add([metadata2_rev])
         self.storage.origin_intrinsic_metadata_add([metadata2_origin])
         self.storage.revision_intrinsic_metadata_add([metadata3_rev])
         self.storage.origin_intrinsic_metadata_add([metadata3_origin])
 
     def test_origin_intrinsic_metadata_search_by_producer(self):
         self._fill_origin_intrinsic_metadata()
         tool1 = self.tools['swh-metadata-detector']
         tool2 = self.tools['swh-metadata-detector2']
         endpoint = self.storage.origin_intrinsic_metadata_search_by_producer
 
         # test pagination
         self.assertCountEqual(
             endpoint(ids_only=True),
             [self.origin_id_1, self.origin_id_2, self.origin_id_3])
         self.assertCountEqual(
             endpoint(start=0, ids_only=True),
             [self.origin_id_1, self.origin_id_2, self.origin_id_3])
         self.assertCountEqual(
             endpoint(start=0, limit=2, ids_only=True),
             [self.origin_id_1, self.origin_id_2])
         self.assertCountEqual(
             endpoint(start=self.origin_id_1+1, ids_only=True),
             [self.origin_id_2, self.origin_id_3])
         self.assertCountEqual(
             endpoint(start=self.origin_id_1+1, end=self.origin_id_3-1,
                      ids_only=True),
             [self.origin_id_2])
 
         # test mappings filtering
         self.assertCountEqual(
             endpoint(mappings=['npm'], ids_only=True),
             [self.origin_id_1, self.origin_id_2])
         self.assertCountEqual(
             endpoint(mappings=['npm', 'gemspec'], ids_only=True),
             [self.origin_id_1, self.origin_id_2])
         self.assertCountEqual(
             endpoint(mappings=['gemspec'], ids_only=True),
             [self.origin_id_2])
         self.assertCountEqual(
             endpoint(mappings=['pkg-info'], ids_only=True),
             [self.origin_id_3])
         self.assertCountEqual(
             endpoint(mappings=['foobar'], ids_only=True),
             [])
 
         # test pagination + mappings
         self.assertCountEqual(
             endpoint(mappings=['npm'], limit=1, ids_only=True),
             [self.origin_id_1])
 
         # test tool filtering
         self.assertCountEqual(
             endpoint(tool_ids=[tool1['id']], ids_only=True),
             [self.origin_id_1])
         self.assertCountEqual(
             endpoint(tool_ids=[tool2['id']], ids_only=True),
             [self.origin_id_2, self.origin_id_3])
         self.assertCountEqual(
             endpoint(tool_ids=[tool1['id'], tool2['id']], ids_only=True),
             [self.origin_id_1, self.origin_id_2, self.origin_id_3])
 
         # test ids_only=False
         self.assertEqual(list(endpoint(mappings=['gemspec'])), [{
             'id': self.origin_id_2,
             'origin_url': 'file:///dev/zero',
             'metadata': {
                 '@context': 'foo',
                 'author': 'Jane Doe',
             },
             'mappings': ['npm', 'gemspec'],
             'tool': tool2,
             'from_revision': self.revision_id_2,
         }])
 
     def test_origin_intrinsic_metadata_stats(self):
         self._fill_origin_intrinsic_metadata()
 
         result = self.storage.origin_intrinsic_metadata_stats()
         self.assertEqual(result, {
             'per_mapping': {
                 'gemspec': 1,
                 'npm': 2,
                 'pkg-info': 1,
                 'codemeta': 0,
                 'maven': 0,
             },
             'total': 3,
             'non_empty': 2,
         })
 
     def test_indexer_configuration_add(self):
         tool = {
             'tool_name': 'some-unknown-tool',
             'tool_version': 'some-version',
             'tool_configuration': {"debian-package": "some-package"},
         }
 
         actual_tool = self.storage.indexer_configuration_get(tool)
         self.assertIsNone(actual_tool)  # does not exist
 
         # add it
         actual_tools = list(self.storage.indexer_configuration_add([tool]))
 
         self.assertEqual(len(actual_tools), 1)
         actual_tool = actual_tools[0]
         self.assertIsNotNone(actual_tool)  # now it exists
         new_id = actual_tool.pop('id')
         self.assertEqual(actual_tool, tool)
 
         actual_tools2 = list(self.storage.indexer_configuration_add([tool]))
         actual_tool2 = actual_tools2[0]
         self.assertIsNotNone(actual_tool2)  # now it exists
         new_id2 = actual_tool2.pop('id')
 
         self.assertEqual(new_id, new_id2)
         self.assertEqual(actual_tool, actual_tool2)
 
     def test_indexer_configuration_add_multiple(self):
         tool = {
             'tool_name': 'some-unknown-tool',
             'tool_version': 'some-version',
             'tool_configuration': {"debian-package": "some-package"},
         }
 
         actual_tools = list(self.storage.indexer_configuration_add([tool]))
         self.assertEqual(len(actual_tools), 1)
 
         new_tools = [tool, {
             'tool_name': 'yet-another-tool',
             'tool_version': 'version',
             'tool_configuration': {},
         }]
 
         actual_tools = list(self.storage.indexer_configuration_add(new_tools))
         self.assertEqual(len(actual_tools), 2)
 
         # order not guaranteed, so we iterate over results to check
         for tool in actual_tools:
             _id = tool.pop('id')
             self.assertIsNotNone(_id)
             self.assertIn(tool, new_tools)
 
     def test_indexer_configuration_get_missing(self):
         tool = {
             'tool_name': 'unknown-tool',
             'tool_version': '3.1.0rc2-31-ga2cbb8c',
             'tool_configuration': {"command_line": "nomossa <filepath>"},
         }
 
         actual_tool = self.storage.indexer_configuration_get(tool)
 
         self.assertIsNone(actual_tool)
 
     def test_indexer_configuration_get(self):
         tool = {
             'tool_name': 'nomos',
             'tool_version': '3.1.0rc2-31-ga2cbb8c',
             'tool_configuration': {"command_line": "nomossa <filepath>"},
         }
 
         self.storage.indexer_configuration_add([tool])
         actual_tool = self.storage.indexer_configuration_get(tool)
 
         expected_tool = tool.copy()
         del actual_tool['id']
 
         self.assertEqual(expected_tool, actual_tool)
 
     def test_indexer_configuration_metadata_get_missing_context(self):
         tool = {
             'tool_name': 'swh-metadata-translator',
             'tool_version': '0.0.1',
             'tool_configuration': {"context": "unknown-context"},
         }
 
         actual_tool = self.storage.indexer_configuration_get(tool)
 
         self.assertIsNone(actual_tool)
 
     def test_indexer_configuration_metadata_get(self):
         tool = {
             'tool_name': 'swh-metadata-translator',
             'tool_version': '0.0.1',
             'tool_configuration': {"type": "local", "context": "NpmMapping"},
         }
 
         self.storage.indexer_configuration_add([tool])
         actual_tool = self.storage.indexer_configuration_get(tool)
 
         expected_tool = tool.copy()
         expected_tool['id'] = actual_tool['id']
 
         self.assertEqual(expected_tool, actual_tool)
 
     @pytest.mark.property_based
     def test_generate_content_mimetype_get_range_limit_none(self):
         """mimetype_get_range call with wrong limit input should fail"""
         with self.assertRaises(ValueError) as e:
             self.storage.content_mimetype_get_range(
                 start=None, end=None, indexer_configuration_id=None,
                 limit=None)
 
         self.assertEqual(e.exception.args, (
             'Development error: limit should not be None',))
 
     @pytest.mark.property_based
     @given(gen_content_mimetypes(min_size=1, max_size=4))
     def test_generate_content_mimetype_get_range_no_limit(self, mimetypes):
         """mimetype_get_range returns mimetypes within range provided"""
         self.reset_storage_tables()
         # add mimetypes to storage
         self.storage.content_mimetype_add(mimetypes)
 
         # All ids from the db
         content_ids = sorted([c['id'] for c in mimetypes])
 
         start = content_ids[0]
         end = content_ids[-1]
 
         # retrieve mimetypes
         tool_id = mimetypes[0]['indexer_configuration_id']
         actual_result = self.storage.content_mimetype_get_range(
             start, end, indexer_configuration_id=tool_id)
 
         actual_ids = actual_result['ids']
         actual_next = actual_result['next']
 
         self.assertEqual(len(mimetypes), len(actual_ids))
         self.assertIsNone(actual_next)
         self.assertEqual(content_ids, actual_ids)
 
     @pytest.mark.property_based
     @given(gen_content_mimetypes(min_size=4, max_size=4))
     def test_generate_content_mimetype_get_range_limit(self, mimetypes):
         """mimetype_get_range paginates results if limit exceeded"""
         self.reset_storage_tables()
 
         # add mimetypes to storage
         self.storage.content_mimetype_add(mimetypes)
 
         # input the list of sha1s we want from storage
         content_ids = sorted([c['id'] for c in mimetypes])
         start = content_ids[0]
         end = content_ids[-1]
 
         # retrieve mimetypes limited to 3 results
         limited_results = len(mimetypes) - 1
         tool_id = mimetypes[0]['indexer_configuration_id']
         actual_result = self.storage.content_mimetype_get_range(
             start, end,
             indexer_configuration_id=tool_id, limit=limited_results)
 
         actual_ids = actual_result['ids']
         actual_next = actual_result['next']
 
         self.assertEqual(limited_results, len(actual_ids))
         self.assertIsNotNone(actual_next)
         self.assertEqual(actual_next, content_ids[-1])
 
         expected_mimetypes = content_ids[:-1]
         self.assertEqual(expected_mimetypes, actual_ids)
 
         # retrieve next part
         actual_results2 = self.storage.content_mimetype_get_range(
             start=end, end=end, indexer_configuration_id=tool_id)
         actual_ids2 = actual_results2['ids']
         actual_next2 = actual_results2['next']
 
         self.assertIsNone(actual_next2)
         expected_mimetypes2 = [content_ids[-1]]
         self.assertEqual(expected_mimetypes2, actual_ids2)
 
     @pytest.mark.property_based
     def test_generate_content_fossology_license_get_range_limit_none(self):
         """license_get_range call with wrong limit input should fail"""
         with self.assertRaises(ValueError) as e:
             self.storage.content_fossology_license_get_range(
                 start=None, end=None, indexer_configuration_id=None,
                 limit=None)
 
         self.assertEqual(e.exception.args, (
             'Development error: limit should not be None',))
 
     @pytest.mark.property_based
     def prepare_mimetypes_from(self, fossology_licenses):
         """Fossology license needs some consistent data in db to run.
 
         """
         mimetypes = []
         for c in fossology_licenses:
             mimetypes.append({
                 'id': c['id'],
                 'mimetype': 'text/plain',
                 'encoding': 'utf-8',
                 'indexer_configuration_id': c['indexer_configuration_id'],
             })
         return mimetypes
 
     @pytest.mark.property_based
     @given(gen_content_fossology_licenses(min_size=1, max_size=4))
     def test_generate_content_fossology_license_get_range_no_limit(
             self, fossology_licenses):
         """license_get_range returns licenses within range provided"""
         self.reset_storage_tables()
         # craft some consistent mimetypes
         mimetypes = self.prepare_mimetypes_from(fossology_licenses)
 
         self.storage.content_mimetype_add(mimetypes)
         # add fossology_licenses to storage
         self.storage.content_fossology_license_add(fossology_licenses)
 
         # All ids from the db
         content_ids = sorted([c['id'] for c in fossology_licenses])
 
         start = content_ids[0]
         end = content_ids[-1]
 
         # retrieve fossology_licenses
         tool_id = fossology_licenses[0]['indexer_configuration_id']
         actual_result = self.storage.content_fossology_license_get_range(
             start, end, indexer_configuration_id=tool_id)
 
         actual_ids = actual_result['ids']
         actual_next = actual_result['next']
 
         self.assertEqual(len(fossology_licenses), len(actual_ids))
         self.assertIsNone(actual_next)
         self.assertEqual(content_ids, actual_ids)
 
     @pytest.mark.property_based
     @given(gen_content_fossology_licenses(min_size=1, max_size=4),
            gen_content_mimetypes(min_size=1, max_size=1))
     def test_generate_content_fossology_license_get_range_no_limit_with_filter(
             self, fossology_licenses, mimetypes):
         """This filters non textual, then returns results within range"""
         self.reset_storage_tables()
 
         # craft some consistent mimetypes
         _mimetypes = self.prepare_mimetypes_from(fossology_licenses)
         # add binary mimetypes which will get filtered out in results
         for m in mimetypes:
             _mimetypes.append({
                 'mimetype': 'binary',
                 **m,
             })
 
         self.storage.content_mimetype_add(_mimetypes)
         # add fossology_licenses to storage
         self.storage.content_fossology_license_add(fossology_licenses)
 
         # All ids from the db
         content_ids = sorted([c['id'] for c in fossology_licenses])
 
         start = content_ids[0]
         end = content_ids[-1]
 
         # retrieve fossology_licenses
         tool_id = fossology_licenses[0]['indexer_configuration_id']
         actual_result = self.storage.content_fossology_license_get_range(
             start, end, indexer_configuration_id=tool_id)
 
         actual_ids = actual_result['ids']
         actual_next = actual_result['next']
 
         self.assertEqual(len(fossology_licenses), len(actual_ids))
         self.assertIsNone(actual_next)
         self.assertEqual(content_ids, actual_ids)
 
     @pytest.mark.property_based
     @given(gen_content_fossology_licenses(min_size=4, max_size=4))
     def test_generate_fossology_license_get_range_limit(
             self, fossology_licenses):
         """fossology_license_get_range paginates results if limit exceeded"""
         self.reset_storage_tables()
         # craft some consistent mimetypes
         mimetypes = self.prepare_mimetypes_from(fossology_licenses)
 
         # add fossology_licenses to storage
         self.storage.content_mimetype_add(mimetypes)
         self.storage.content_fossology_license_add(fossology_licenses)
 
         # input the list of sha1s we want from storage
         content_ids = sorted([c['id'] for c in fossology_licenses])
         start = content_ids[0]
         end = content_ids[-1]
 
         # retrieve fossology_licenses limited to 3 results
         limited_results = len(fossology_licenses) - 1
         tool_id = fossology_licenses[0]['indexer_configuration_id']
         actual_result = self.storage.content_fossology_license_get_range(
             start, end,
             indexer_configuration_id=tool_id, limit=limited_results)
 
         actual_ids = actual_result['ids']
         actual_next = actual_result['next']
 
         self.assertEqual(limited_results, len(actual_ids))
         self.assertIsNotNone(actual_next)
         self.assertEqual(actual_next, content_ids[-1])
 
         expected_fossology_licenses = content_ids[:-1]
         self.assertEqual(expected_fossology_licenses, actual_ids)
 
         # retrieve next part
         actual_results2 = self.storage.content_fossology_license_get_range(
             start=end, end=end, indexer_configuration_id=tool_id)
         actual_ids2 = actual_results2['ids']
         actual_next2 = actual_results2['next']
 
         self.assertIsNone(actual_next2)
         expected_fossology_licenses2 = [content_ids[-1]]
         self.assertEqual(expected_fossology_licenses2, actual_ids2)
 
 
 @pytest.mark.db
 class IndexerTestStorage(CommonTestStorage, BasePgTestStorage,
                          unittest.TestCase):
     """Running the tests locally.
 
     For the client api tests (remote storage), see
     `class`:swh.indexer.storage.test_api_client:TestRemoteStorage
     class.
 
     """
     pass
 
 
 def test_mapping_names():
     assert set(MAPPING_NAMES) == {m.name for m in MAPPINGS.values()}
diff --git a/swh/indexer/tests/test_fossology_license.py b/swh/indexer/tests/test_fossology_license.py
index 75803ed..cd6030b 100644
--- a/swh/indexer/tests/test_fossology_license.py
+++ b/swh/indexer/tests/test_fossology_license.py
@@ -1,180 +1,181 @@
 # Copyright (C) 2017-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import unittest
-from unittest.mock import patch
-
 import pytest
 
+from unittest.mock import patch
+from typing import Any, Dict
+
 from swh.indexer import fossology_license
 from swh.indexer.fossology_license import (
     FossologyLicenseIndexer, FossologyLicenseRangeIndexer,
     compute_license
 )
 
 from swh.indexer.tests.utils import (
     SHA1_TO_LICENSES, CommonContentIndexerTest, CommonContentIndexerRangeTest,
     BASE_TEST_CONFIG, fill_storage, fill_obj_storage, filter_dict,
 )
 
 
 class BasicTest(unittest.TestCase):
     @patch('swh.indexer.fossology_license.subprocess')
     def test_compute_license(self, mock_subprocess):
         """Computing licenses from a raw content should return results
 
         """
         for path, intermediary_result, output in [
                 (b'some/path', None,
                  []),
                 (b'some/path/2', [],
                  []),
                 (b'other/path', ' contains license(s) GPL,AGPL',
                  ['GPL', 'AGPL'])]:
             mock_subprocess.check_output.return_value = intermediary_result
 
             actual_result = compute_license(path, log=None)
 
             self.assertEqual(actual_result, {
                 'licenses': output,
                 'path': path,
             })
 
 
 def mock_compute_license(path, log=None):
     """path is the content identifier
 
     """
     if isinstance(id, bytes):
         path = path.decode('utf-8')
     # path is something like /tmp/tmpXXX/<sha1> so we keep only the sha1 part
     path = path.split('/')[-1]
     return {
         'licenses': SHA1_TO_LICENSES.get(path)
     }
 
 
 CONFIG = {
     **BASE_TEST_CONFIG,
     'workdir': '/tmp',
     'tools': {
         'name': 'nomos',
         'version': '3.1.0rc2-31-ga2cbb8c',
         'configuration': {
             'command_line': 'nomossa <filepath>',
         },
     },
-}
+}  # type: Dict[str, Any]
 
 RANGE_CONFIG = dict(list(CONFIG.items()) + [('write_batch_size', 100)])
 
 
 class TestFossologyLicenseIndexer(CommonContentIndexerTest, unittest.TestCase):
     """Language indexer test scenarios:
 
     - Known sha1s in the input list have their data indexed
     - Unknown sha1 in the input list are not indexed
 
     """
 
     def get_indexer_results(self, ids):
         yield from self.idx_storage.content_fossology_license_get(ids)
 
     def setUp(self):
         super().setUp()
         # replace actual license computation with a mock
         self.orig_compute_license = fossology_license.compute_license
         fossology_license.compute_license = mock_compute_license
 
         self.indexer = FossologyLicenseIndexer(CONFIG)
         self.indexer.catch_exceptions = False
         self.idx_storage = self.indexer.idx_storage
         fill_storage(self.indexer.storage)
         fill_obj_storage(self.indexer.objstorage)
 
         self.id0 = '01c9379dfc33803963d07c1ccc748d3fe4c96bb5'
         self.id1 = '688a5ef812c53907562fe379d4b3851e69c7cb15'
         self.id2 = 'da39a3ee5e6b4b0d3255bfef95601890afd80709'  # empty content
 
         tool = {k.replace('tool_', ''): v
                 for (k, v) in self.indexer.tool.items()}
         # then
         self.expected_results = {
             self.id0: {
                 'tool': tool,
                 'licenses': SHA1_TO_LICENSES[self.id0],
             },
             self.id1: {
                 'tool': tool,
                 'licenses': SHA1_TO_LICENSES[self.id1],
             },
             self.id2: {
                 'tool': tool,
                 'licenses': SHA1_TO_LICENSES[self.id2],
             }
         }
 
     def tearDown(self):
         super().tearDown()
         fossology_license.compute_license = self.orig_compute_license
 
 
 class TestFossologyLicenseRangeIndexer(
         CommonContentIndexerRangeTest, unittest.TestCase):
     """Range Fossology License Indexer tests.
 
     - new data within range are indexed
     - no data outside a range are indexed
     - with filtering existing indexed data prior to compute new index
     - without filtering existing indexed data prior to compute new index
 
     """
     def setUp(self):
         super().setUp()
 
         # replace actual license computation with a mock
         self.orig_compute_license = fossology_license.compute_license
         fossology_license.compute_license = mock_compute_license
 
         self.indexer = FossologyLicenseRangeIndexer(config=RANGE_CONFIG)
         self.indexer.catch_exceptions = False
         fill_storage(self.indexer.storage)
         fill_obj_storage(self.indexer.objstorage)
 
         self.id0 = '01c9379dfc33803963d07c1ccc748d3fe4c96bb5'
         self.id1 = '02fb2c89e14f7fab46701478c83779c7beb7b069'
         self.id2 = '103bc087db1d26afc3a0283f38663d081e9b01e6'
         tool_id = self.indexer.tool['id']
         self.expected_results = {
             self.id0: {
                 'id': self.id0,
                 'indexer_configuration_id': tool_id,
                 'licenses': SHA1_TO_LICENSES[self.id0]
             },
             self.id1: {
                 'id': self.id1,
                 'indexer_configuration_id': tool_id,
                 'licenses': SHA1_TO_LICENSES[self.id1]
             },
             self.id2: {
                 'id': self.id2,
                 'indexer_configuration_id': tool_id,
                 'licenses': SHA1_TO_LICENSES[self.id2]
             }
         }
 
     def tearDown(self):
         super().tearDown()
         fossology_license.compute_license = self.orig_compute_license
 
 
 def test_fossology_w_no_tool():
     with pytest.raises(ValueError):
         FossologyLicenseIndexer(config=filter_dict(CONFIG, 'tools'))
 
 
 def test_fossology_range_w_no_tool():
     with pytest.raises(ValueError):
         FossologyLicenseRangeIndexer(config=filter_dict(RANGE_CONFIG, 'tools'))
diff --git a/swh/indexer/tests/test_metadata.py b/swh/indexer/tests/test_metadata.py
index 045f4de..6eb5cbb 100644
--- a/swh/indexer/tests/test_metadata.py
+++ b/swh/indexer/tests/test_metadata.py
@@ -1,1207 +1,1212 @@
 # Copyright (C) 2017-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import json
 import unittest
 
 from hypothesis import given, strategies, settings, HealthCheck
+from typing import cast
 
 from swh.model.hashutil import hash_to_bytes
 
 from swh.indexer.codemeta import CODEMETA_TERMS, CROSSWALK_TABLE
 from swh.indexer.codemeta import merge_documents
 from swh.indexer.metadata_dictionary import MAPPINGS
 from swh.indexer.metadata_dictionary.base import merge_values
+from swh.indexer.metadata_dictionary.maven import MavenMapping
+from swh.indexer.metadata_dictionary.npm import NpmMapping
+from swh.indexer.metadata_dictionary.ruby import GemspecMapping
 from swh.indexer.metadata_detector import (
     detect_metadata
 )
 from swh.indexer.metadata import (
     ContentMetadataIndexer, RevisionMetadataIndexer
 )
 
 from .utils import (
     BASE_TEST_CONFIG, fill_obj_storage, fill_storage,
     YARN_PARSER_METADATA, json_document_strategy,
     xml_document_strategy,
 )
 
 
 TRANSLATOR_TOOL = {
     'name': 'swh-metadata-translator',
     'version': '0.0.2',
     'configuration': {
         'type': 'local',
         'context': 'NpmMapping'
     }
 }
 
 
 class ContentMetadataTestIndexer(ContentMetadataIndexer):
     """Specific Metadata whose configuration is enough to satisfy the
        indexing tests.
     """
     def parse_config_file(self, *args, **kwargs):
         assert False, 'should not be called; the rev indexer configures it.'
 
 
 REVISION_METADATA_CONFIG = {
     **BASE_TEST_CONFIG,
     'tools': TRANSLATOR_TOOL,
 }
 
 
 class Metadata(unittest.TestCase):
     """
     Tests metadata_mock_tool tool for Metadata detection
     """
     def setUp(self):
         """
         shows the entire diff in the results
         """
         self.maxDiff = None
         self.npm_mapping = MAPPINGS['NpmMapping']()
         self.codemeta_mapping = MAPPINGS['CodemetaMapping']()
         self.maven_mapping = MAPPINGS['MavenMapping']()
         self.pkginfo_mapping = MAPPINGS['PythonPkginfoMapping']()
         self.gemspec_mapping = MAPPINGS['GemspecMapping']()
 
     def test_crosstable(self):
         self.assertEqual(CROSSWALK_TABLE['NodeJS'], {
             'repository': 'http://schema.org/codeRepository',
             'os': 'http://schema.org/operatingSystem',
             'cpu': 'http://schema.org/processorRequirements',
             'engines':
                 'http://schema.org/processorRequirements',
             'author': 'http://schema.org/author',
             'author.email': 'http://schema.org/email',
             'author.name': 'http://schema.org/name',
             'contributor': 'http://schema.org/contributor',
             'keywords': 'http://schema.org/keywords',
             'license': 'http://schema.org/license',
             'version': 'http://schema.org/version',
             'description': 'http://schema.org/description',
             'name': 'http://schema.org/name',
             'bugs': 'https://codemeta.github.io/terms/issueTracker',
             'homepage': 'http://schema.org/url'
         })
 
     def test_merge_values(self):
         self.assertEqual(
             merge_values('a', 'b'),
             ['a', 'b'])
         self.assertEqual(
             merge_values(['a', 'b'], 'c'),
             ['a', 'b', 'c'])
         self.assertEqual(
             merge_values('a', ['b', 'c']),
             ['a', 'b', 'c'])
 
         self.assertEqual(
             merge_values({'@list': ['a']}, {'@list': ['b']}),
             {'@list': ['a', 'b']})
         self.assertEqual(
             merge_values({'@list': ['a', 'b']}, {'@list': ['c']}),
             {'@list': ['a', 'b', 'c']})
 
         with self.assertRaises(ValueError):
             merge_values({'@list': ['a']}, 'b')
         with self.assertRaises(ValueError):
             merge_values('a', {'@list': ['b']})
         with self.assertRaises(ValueError):
             merge_values({'@list': ['a']}, ['b'])
         with self.assertRaises(ValueError):
             merge_values(['a'], {'@list': ['b']})
 
         self.assertEqual(
             merge_values('a', None),
             'a')
         self.assertEqual(
             merge_values(['a', 'b'], None),
             ['a', 'b'])
         self.assertEqual(
             merge_values(None, ['b', 'c']),
             ['b', 'c'])
         self.assertEqual(
             merge_values({'@list': ['a']}, None),
             {'@list': ['a']})
         self.assertEqual(
             merge_values(None, {'@list': ['a']}),
             {'@list': ['a']})
 
     def test_compute_metadata_none(self):
         """
         testing content empty content is empty
         should return None
         """
         # given
         content = b""
 
         # None if no metadata was found or an error occurred
         declared_metadata = None
         # when
         result = self.npm_mapping.translate(content)
         # then
         self.assertEqual(declared_metadata, result)
 
     def test_compute_metadata_npm(self):
         """
         testing only computation of metadata with hard_mapping_npm
         """
         # given
         content = b"""
             {
                 "name": "test_metadata",
                 "version": "0.0.2",
                 "description": "Simple package.json test for indexer",
                   "repository": {
                     "type": "git",
                     "url": "https://github.com/moranegg/metadata_test"
                 },
                 "author": {
                     "email": "moranegg@example.com",
                     "name": "Morane G"
                 }
             }
         """
         declared_metadata = {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'test_metadata',
             'version': '0.0.2',
             'description': 'Simple package.json test for indexer',
             'codeRepository':
                 'git+https://github.com/moranegg/metadata_test',
             'author': [{
                 'type': 'Person',
                 'name': 'Morane G',
                 'email': 'moranegg@example.com',
             }],
         }
 
         # when
         result = self.npm_mapping.translate(content)
         # then
         self.assertEqual(declared_metadata, result)
 
     def test_merge_documents(self):
         """
         Test the creation of a coherent minimal metadata set
         """
         # given
         metadata_list = [{
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'test_1',
             'version': '0.0.2',
             'description': 'Simple package.json test for indexer',
             'codeRepository':
                 'git+https://github.com/moranegg/metadata_test',
         }, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'test_0_1',
             'version': '0.0.2',
             'description': 'Simple package.json test for indexer',
             'codeRepository':
                 'git+https://github.com/moranegg/metadata_test'
         }, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'test_metadata',
             'version': '0.0.2',
             'author': 'moranegg',
         }]
 
         # when
         results = merge_documents(metadata_list)
 
         # then
         expected_results = {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             "version": '0.0.2',
             "description": 'Simple package.json test for indexer',
             "name": ['test_1', 'test_0_1', 'test_metadata'],
             "author": ['moranegg'],
             "codeRepository":
                 'git+https://github.com/moranegg/metadata_test',
         }
         self.assertEqual(expected_results, results)
 
     def test_index_content_metadata_npm(self):
         """
         testing NPM with package.json
         - one sha1 uses a file that can't be translated to metadata and
           should return None in the translated metadata
         """
         # given
         sha1s = [
             hash_to_bytes('26a9f72a7c87cc9205725cfd879f514ff4f3d8d5'),
             hash_to_bytes('d4c647f0fc257591cc9ba1722484229780d1c607'),
             hash_to_bytes('02fb2c89e14f7fab46701478c83779c7beb7b069'),
         ]
         # this metadata indexer computes only metadata for package.json
         # in npm context with a hard mapping
         config = BASE_TEST_CONFIG.copy()
         config['tools'] = [TRANSLATOR_TOOL]
         metadata_indexer = ContentMetadataTestIndexer(config=config)
         fill_obj_storage(metadata_indexer.objstorage)
         fill_storage(metadata_indexer.storage)
 
         # when
         metadata_indexer.run(sha1s, policy_update='ignore-dups')
         results = list(metadata_indexer.idx_storage.content_metadata_get(
             sha1s))
 
         expected_results = [{
             'metadata': {
                 '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
                 'type': 'SoftwareSourceCode',
                 'codeRepository':
                     'git+https://github.com/moranegg/metadata_test',
                 'description': 'Simple package.json test for indexer',
                 'name': 'test_metadata',
                 'version': '0.0.1'
             },
             'id': hash_to_bytes('26a9f72a7c87cc9205725cfd879f514ff4f3d8d5'),
             }, {
             'metadata': {
                 '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
                 'type': 'SoftwareSourceCode',
                 'issueTracker':
                     'https://github.com/npm/npm/issues',
                 'author': [{
                     'type': 'Person',
                     'name': 'Isaac Z. Schlueter',
                     'email': 'i@izs.me',
                     'url': 'http://blog.izs.me',
                 }],
                 'codeRepository':
                     'git+https://github.com/npm/npm',
                 'description': 'a package manager for JavaScript',
                 'license': 'https://spdx.org/licenses/Artistic-2.0',
                 'version': '5.0.3',
                 'name': 'npm',
                 'keywords': [
                     'install',
                     'modules',
                     'package manager',
                     'package.json'
                 ],
                 'url': 'https://docs.npmjs.com/'
             },
             'id': hash_to_bytes('d4c647f0fc257591cc9ba1722484229780d1c607')
         }]
 
         for result in results:
             del result['tool']
 
         # The assertion below returns False sometimes because of nested lists
         self.assertEqual(expected_results, results)
 
     def test_npm_bugs_normalization(self):
         # valid dictionary
         package_json = b"""{
             "name": "foo",
             "bugs": {
                 "url": "https://github.com/owner/project/issues",
                 "email": "foo@example.com"
             }
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'issueTracker': 'https://github.com/owner/project/issues',
             'type': 'SoftwareSourceCode',
         })
 
         # "invalid" dictionary
         package_json = b"""{
             "name": "foo",
             "bugs": {
                 "email": "foo@example.com"
             }
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'type': 'SoftwareSourceCode',
         })
 
         # string
         package_json = b"""{
             "name": "foo",
             "bugs": "https://github.com/owner/project/issues"
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'issueTracker': 'https://github.com/owner/project/issues',
             'type': 'SoftwareSourceCode',
         })
 
     def test_npm_repository_normalization(self):
         # normal
         package_json = b"""{
             "name": "foo",
             "repository": {
                 "type" : "git",
                 "url" : "https://github.com/npm/cli.git"
             }
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'codeRepository': 'git+https://github.com/npm/cli.git',
             'type': 'SoftwareSourceCode',
         })
 
         # missing url
         package_json = b"""{
             "name": "foo",
             "repository": {
                 "type" : "git"
             }
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'type': 'SoftwareSourceCode',
         })
 
         # github shortcut
         package_json = b"""{
             "name": "foo",
             "repository": "github:npm/cli"
         }"""
         result = self.npm_mapping.translate(package_json)
         expected_result = {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'codeRepository': 'git+https://github.com/npm/cli.git',
             'type': 'SoftwareSourceCode',
         }
         self.assertEqual(result, expected_result)
 
         # github shortshortcut
         package_json = b"""{
             "name": "foo",
             "repository": "npm/cli"
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, expected_result)
 
         # gitlab shortcut
         package_json = b"""{
             "name": "foo",
             "repository": "gitlab:user/repo"
         }"""
         result = self.npm_mapping.translate(package_json)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'name': 'foo',
             'codeRepository': 'git+https://gitlab.com/user/repo.git',
             'type': 'SoftwareSourceCode',
         })
 
     def test_detect_metadata_package_json(self):
         # given
         df = [{
                 'sha1_git': b'abc',
                 'name': b'index.js',
                 'target': b'abc',
                 'length': 897,
                 'status': 'visible',
                 'type': 'file',
                 'perms': 33188,
                 'dir_id': b'dir_a',
                 'sha1': b'bcd'
             },
             {
                 'sha1_git': b'aab',
                 'name': b'package.json',
                 'target': b'aab',
                 'length': 712,
                 'status': 'visible',
                 'type': 'file',
                 'perms': 33188,
                 'dir_id': b'dir_a',
                 'sha1': b'cde'
         }]
         # when
         results = detect_metadata(df)
 
         expected_results = {
             'NpmMapping': [
                 b'cde'
             ]
         }
         # then
         self.assertEqual(expected_results, results)
 
     def test_compute_metadata_valid_codemeta(self):
         raw_content = (
             b"""{
             "@context": "https://doi.org/10.5063/schema/codemeta-2.0",
             "@type": "SoftwareSourceCode",
             "identifier": "CodeMeta",
             "description": "CodeMeta is a concept vocabulary that can be used to standardize the exchange of software metadata across repositories and organizations.",
             "name": "CodeMeta: Minimal metadata schemas for science software and code, in JSON-LD",
             "codeRepository": "https://github.com/codemeta/codemeta",
             "issueTracker": "https://github.com/codemeta/codemeta/issues",
             "license": "https://spdx.org/licenses/Apache-2.0",
             "version": "2.0",
             "author": [
               {
                 "@type": "Person",
                 "givenName": "Carl",
                 "familyName": "Boettiger",
                 "email": "cboettig@gmail.com",
                 "@id": "http://orcid.org/0000-0002-1642-628X"
               },
               {
                 "@type": "Person",
                 "givenName": "Matthew B.",
                 "familyName": "Jones",
                 "email": "jones@nceas.ucsb.edu",
                 "@id": "http://orcid.org/0000-0003-0077-4738"
               }
             ],
             "maintainer": {
               "@type": "Person",
               "givenName": "Carl",
               "familyName": "Boettiger",
               "email": "cboettig@gmail.com",
               "@id": "http://orcid.org/0000-0002-1642-628X"
             },
             "contIntegration": "https://travis-ci.org/codemeta/codemeta",
             "developmentStatus": "active",
             "downloadUrl": "https://github.com/codemeta/codemeta/archive/2.0.zip",
             "funder": {
                 "@id": "https://doi.org/10.13039/100000001",
                 "@type": "Organization",
                 "name": "National Science Foundation"
             },
             "funding":"1549758; Codemeta: A Rosetta Stone for Metadata in Scientific Software",
             "keywords": [
               "metadata",
               "software"
             ],
             "version":"2.0",
             "dateCreated":"2017-06-05",
             "datePublished":"2017-06-05",
             "programmingLanguage": "JSON-LD"
           }""") # noqa
         expected_result = {
             "@context": "https://doi.org/10.5063/schema/codemeta-2.0",
             "type": "SoftwareSourceCode",
             "identifier": "CodeMeta",
             "description":
                 "CodeMeta is a concept vocabulary that can "
                 "be used to standardize the exchange of software metadata "
                 "across repositories and organizations.",
             "name":
                 "CodeMeta: Minimal metadata schemas for science "
                 "software and code, in JSON-LD",
             "codeRepository": "https://github.com/codemeta/codemeta",
             "issueTracker": "https://github.com/codemeta/codemeta/issues",
             "license": "https://spdx.org/licenses/Apache-2.0",
             "version": "2.0",
             "author": [
               {
                 "type": "Person",
                 "givenName": "Carl",
                 "familyName": "Boettiger",
                 "email": "cboettig@gmail.com",
                 "id": "http://orcid.org/0000-0002-1642-628X"
               },
               {
                 "type": "Person",
                 "givenName": "Matthew B.",
                 "familyName": "Jones",
                 "email": "jones@nceas.ucsb.edu",
                 "id": "http://orcid.org/0000-0003-0077-4738"
               }
             ],
             "maintainer": {
               "type": "Person",
               "givenName": "Carl",
               "familyName": "Boettiger",
               "email": "cboettig@gmail.com",
               "id": "http://orcid.org/0000-0002-1642-628X"
             },
             "contIntegration": "https://travis-ci.org/codemeta/codemeta",
             "developmentStatus": "active",
             "downloadUrl":
                 "https://github.com/codemeta/codemeta/archive/2.0.zip",
             "funder": {
                 "id": "https://doi.org/10.13039/100000001",
                 "type": "Organization",
                 "name": "National Science Foundation"
             },
             "funding": "1549758; Codemeta: A Rosetta Stone for Metadata "
                 "in Scientific Software",
             "keywords": [
               "metadata",
               "software"
             ],
             "version": "2.0",
             "dateCreated": "2017-06-05",
             "datePublished": "2017-06-05",
             "programmingLanguage": "JSON-LD"
           }
         result = self.codemeta_mapping.translate(raw_content)
         self.assertEqual(result, expected_result)
 
     def test_compute_metadata_codemeta_alternate_context(self):
         raw_content = (
             b"""{
             "@context": "https://raw.githubusercontent.com/codemeta/codemeta/master/codemeta.jsonld",
             "@type": "SoftwareSourceCode",
             "identifier": "CodeMeta"
         }""")  # noqa
         expected_result = {
             "@context": "https://doi.org/10.5063/schema/codemeta-2.0",
             "type": "SoftwareSourceCode",
             "identifier": "CodeMeta",
         }
         result = self.codemeta_mapping.translate(raw_content)
         self.assertEqual(result, expected_result)
 
     def test_compute_metadata_maven(self):
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
           <repositories>
             <repository>
               <id>central</id>
               <name>Maven Repository Switchboard</name>
               <layout>default</layout>
               <url>http://repo1.maven.org/maven2</url>
               <snapshots>
                 <enabled>false</enabled>
               </snapshots>
             </repository>
           </repositories>
           <licenses>
             <license>
               <name>Apache License, Version 2.0</name>
               <url>https://www.apache.org/licenses/LICENSE-2.0.txt</url>
               <distribution>repo</distribution>
               <comments>A business-friendly OSS license</comments>
             </license>
           </licenses>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'license': 'https://www.apache.org/licenses/LICENSE-2.0.txt',
             'codeRepository':
                 'http://repo1.maven.org/maven2/com/mycompany/app/my-app',
         })
 
     def test_compute_metadata_maven_empty(self):
         raw_content = b"""
         <project>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
         })
 
     def test_compute_metadata_maven_almost_empty(self):
         raw_content = b"""
         <project>
           <foo/>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
         })
 
     def test_compute_metadata_maven_invalid_xml(self):
         expected_warning = (
             'WARNING:swh.indexer.metadata_dictionary.maven.MavenMapping:'
             'Error parsing XML from foo')
 
         raw_content = b"""
         <project>"""
         with self.assertLogs('swh.indexer.metadata_dictionary',
                              level='WARNING') as cm:
             result = MAPPINGS["MavenMapping"]('foo').translate(raw_content)
             self.assertEqual(cm.output, [expected_warning])
         self.assertEqual(result, None)
 
         raw_content = b"""
         """
         with self.assertLogs('swh.indexer.metadata_dictionary',
                              level='WARNING') as cm:
             result = MAPPINGS["MavenMapping"]('foo').translate(raw_content)
             self.assertEqual(cm.output, [expected_warning])
         self.assertEqual(result, None)
 
     def test_compute_metadata_maven_unknown_encoding(self):
         expected_warning = (
             'WARNING:swh.indexer.metadata_dictionary.maven.MavenMapping:'
             'Error detecting XML encoding from foo')
 
         raw_content = b"""<?xml version="1.0" encoding="foo"?>
         <project>
         </project>"""
         with self.assertLogs('swh.indexer.metadata_dictionary',
                              level='WARNING') as cm:
             result = MAPPINGS["MavenMapping"]('foo').translate(raw_content)
             self.assertEqual(cm.output, [expected_warning])
         self.assertEqual(result, None)
 
         raw_content = b"""<?xml version="1.0" encoding="UTF-7"?>
         <project>
         </project>"""
         with self.assertLogs('swh.indexer.metadata_dictionary',
                              level='WARNING') as cm:
             result = MAPPINGS["MavenMapping"]('foo').translate(raw_content)
             self.assertEqual(cm.output, [expected_warning])
         self.assertEqual(result, None)
 
     def test_compute_metadata_maven_invalid_encoding(self):
         expected_warning = (
             'WARNING:swh.indexer.metadata_dictionary.maven.MavenMapping:'
             'Error unidecoding XML from foo')
 
         raw_content = b"""<?xml version="1.0" encoding="UTF-8"?>
         <foo\xe5ct>
         </foo>"""
         with self.assertLogs('swh.indexer.metadata_dictionary',
                              level='WARNING') as cm:
             result = MAPPINGS["MavenMapping"]('foo').translate(raw_content)
             self.assertEqual(cm.output, [expected_warning])
         self.assertEqual(result, None)
 
     def test_compute_metadata_maven_minimal(self):
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'codeRepository':
             'https://repo.maven.apache.org/maven2/com/mycompany/app/my-app',
         })
 
     def test_compute_metadata_maven_empty_nodes(self):
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
           <repositories>
           </repositories>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'codeRepository':
             'https://repo.maven.apache.org/maven2/com/mycompany/app/my-app',
         })
 
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version></version>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'codeRepository':
             'https://repo.maven.apache.org/maven2/com/mycompany/app/my-app',
         })
 
         raw_content = b"""
         <project>
           <name></name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'codeRepository':
             'https://repo.maven.apache.org/maven2/com/mycompany/app/my-app',
         })
 
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
           <licenses>
           </licenses>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'codeRepository':
             'https://repo.maven.apache.org/maven2/com/mycompany/app/my-app',
         })
 
         raw_content = b"""
         <project>
           <groupId></groupId>
           <version>1.2.3</version>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'version': '1.2.3',
         })
 
     def test_compute_metadata_maven_invalid_licenses(self):
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
           <licenses>
             foo
           </licenses>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'codeRepository':
             'https://repo.maven.apache.org/maven2/com/mycompany/app/my-app',
         })
 
     def test_compute_metadata_maven_multiple(self):
         '''Tests when there are multiple code repos and licenses.'''
         raw_content = b"""
         <project>
           <name>Maven Default Project</name>
           <modelVersion>4.0.0</modelVersion>
           <groupId>com.mycompany.app</groupId>
           <artifactId>my-app</artifactId>
           <version>1.2.3</version>
           <repositories>
             <repository>
               <id>central</id>
               <name>Maven Repository Switchboard</name>
               <layout>default</layout>
               <url>http://repo1.maven.org/maven2</url>
               <snapshots>
                 <enabled>false</enabled>
               </snapshots>
             </repository>
             <repository>
               <id>example</id>
               <name>Example Maven Repo</name>
               <layout>default</layout>
               <url>http://example.org/maven2</url>
             </repository>
           </repositories>
           <licenses>
             <license>
               <name>Apache License, Version 2.0</name>
               <url>https://www.apache.org/licenses/LICENSE-2.0.txt</url>
               <distribution>repo</distribution>
               <comments>A business-friendly OSS license</comments>
             </license>
             <license>
               <name>MIT license</name>
               <url>https://opensource.org/licenses/MIT</url>
             </license>
           </licenses>
         </project>"""
         result = self.maven_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'Maven Default Project',
             'identifier': 'com.mycompany.app',
             'version': '1.2.3',
             'license': [
                 'https://www.apache.org/licenses/LICENSE-2.0.txt',
                 'https://opensource.org/licenses/MIT',
             ],
             'codeRepository': [
                 'http://repo1.maven.org/maven2/com/mycompany/app/my-app',
                 'http://example.org/maven2/com/mycompany/app/my-app',
             ]
         })
 
     def test_compute_metadata_pkginfo(self):
         raw_content = (b"""\
 Metadata-Version: 2.1
 Name: swh.core
 Version: 0.0.49
 Summary: Software Heritage core utilities
 Home-page: https://forge.softwareheritage.org/diffusion/DCORE/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 License: UNKNOWN
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-core
 Description: swh-core
         ========
        \x20
         core library for swh's modules:
         - config parser
         - hash computations
         - serialization
         - logging mechanism
        \x20
 Platform: UNKNOWN
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Description-Content-Type: text/markdown
 Provides-Extra: testing
 """) # noqa
         result = self.pkginfo_mapping.translate(raw_content)
         self.assertCountEqual(result['description'], [
             'Software Heritage core utilities',  # note the comma here
             'swh-core\n'
             '========\n'
             '\n'
             "core library for swh's modules:\n"
             '- config parser\n'
             '- hash computations\n'
             '- serialization\n'
             '- logging mechanism\n'
             ''],
             result)
         del result['description']
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'url': 'https://forge.softwareheritage.org/diffusion/DCORE/',
             'name': 'swh.core',
             'author': [{
                 'type': 'Person',
                 'name': 'Software Heritage developers',
                 'email': 'swh-devel@inria.fr',
             }],
             'version': '0.0.49',
         })
 
     def test_compute_metadata_pkginfo_utf8(self):
         raw_content = (b'''\
 Metadata-Version: 1.1
 Name: snowpyt
 Description-Content-Type: UNKNOWN
 Description: foo
         Hydrology N\xc2\xb083
 ''') # noqa
         result = self.pkginfo_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'snowpyt',
             'description': 'foo\nHydrology N°83',
         })
 
     def test_compute_metadata_pkginfo_keywords(self):
         raw_content = (b"""\
 Metadata-Version: 2.1
 Name: foo
 Keywords: foo bar baz
 """) # noqa
         result = self.pkginfo_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'foo',
             'keywords': ['foo', 'bar', 'baz'],
         })
 
     def test_compute_metadata_pkginfo_license(self):
         raw_content = (b"""\
 Metadata-Version: 2.1
 Name: foo
 License: MIT
 """) # noqa
         result = self.pkginfo_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'foo',
             'license': 'MIT',
         })
 
     def test_gemspec_base(self):
         raw_content = b"""
 Gem::Specification.new do |s|
   s.name        = 'example'
   s.version     = '0.1.0'
   s.licenses    = ['MIT']
   s.summary     = "This is an example!"
   s.description = "Much longer explanation of the example!"
   s.authors     = ["Ruby Coder"]
   s.email       = 'rubycoder@example.com'
   s.files       = ["lib/example.rb"]
   s.homepage    = 'https://rubygems.org/gems/example'
   s.metadata    = { "source_code_uri" => "https://github.com/example/example" }
 end"""
         result = self.gemspec_mapping.translate(raw_content)
         self.assertCountEqual(result.pop('description'), [
             "This is an example!",
             "Much longer explanation of the example!"
         ])
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'author': ['Ruby Coder'],
             'name': 'example',
             'license': 'https://spdx.org/licenses/MIT',
             'codeRepository': 'https://rubygems.org/gems/example',
             'email': 'rubycoder@example.com',
             'version': '0.1.0',
         })
 
     def test_gemspec_two_author_fields(self):
         raw_content = b"""
 Gem::Specification.new do |s|
   s.authors     = ["Ruby Coder1"]
   s.author      = "Ruby Coder2"
 end"""
         result = self.gemspec_mapping.translate(raw_content)
         self.assertCountEqual(result.pop('author'), [
             'Ruby Coder1', 'Ruby Coder2'])
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
         })
 
     def test_gemspec_invalid_author(self):
         raw_content = b"""
 Gem::Specification.new do |s|
   s.author      = ["Ruby Coder"]
 end"""
         result = self.gemspec_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
         })
         raw_content = b"""
 Gem::Specification.new do |s|
   s.author      = "Ruby Coder1",
 end"""
         result = self.gemspec_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
         })
         raw_content = b"""
 Gem::Specification.new do |s|
   s.authors     = ["Ruby Coder1", ["Ruby Coder2"]]
 end"""
         result = self.gemspec_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'author': ['Ruby Coder1'],
         })
 
     def test_gemspec_alternative_header(self):
         raw_content = b"""
 require './lib/version'
 
 Gem::Specification.new { |s|
   s.name = 'rb-system-with-aliases'
   s.summary = 'execute system commands with aliases'
 }
 """
         result = self.gemspec_mapping.translate(raw_content)
         self.assertEqual(result, {
             '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
             'type': 'SoftwareSourceCode',
             'name': 'rb-system-with-aliases',
             'description': 'execute system commands with aliases',
         })
 
     @settings(suppress_health_check=[HealthCheck.too_slow])
     @given(json_document_strategy(
-        keys=list(MAPPINGS['NpmMapping'].mapping)))
+        keys=list(cast(NpmMapping, MAPPINGS['NpmMapping']).mapping)))
     def test_npm_adversarial(self, doc):
         raw = json.dumps(doc).encode()
         self.npm_mapping.translate(raw)
 
     @settings(suppress_health_check=[HealthCheck.too_slow])
     @given(json_document_strategy(keys=CODEMETA_TERMS))
     def test_codemeta_adversarial(self, doc):
         raw = json.dumps(doc).encode()
         self.codemeta_mapping.translate(raw)
 
     @settings(suppress_health_check=[HealthCheck.too_slow])
     @given(xml_document_strategy(
-        keys=list(MAPPINGS['MavenMapping'].mapping),
+        keys=list(cast(MavenMapping, MAPPINGS['MavenMapping']).mapping),
         root='project',
         xmlns='http://maven.apache.org/POM/4.0.0'))
     def test_maven_adversarial(self, doc):
         self.maven_mapping.translate(doc)
 
     @settings(suppress_health_check=[HealthCheck.too_slow])
     @given(strategies.dictionaries(
         # keys
         strategies.one_of(
             strategies.text(),
-            *map(strategies.just, MAPPINGS['GemspecMapping'].mapping)
+            *map(strategies.just,
+                 cast(GemspecMapping, MAPPINGS['GemspecMapping']).mapping)
         ),
         # values
         strategies.recursive(
             strategies.characters(),
             lambda children: strategies.lists(children, 1)
         )
     ))
     def test_gemspec_adversarial(self, doc):
         parts = [b'Gem::Specification.new do |s|\n']
         for (k, v) in doc.items():
             parts.append('  s.{} = {}\n'.format(k, repr(v)).encode())
         parts.append(b'end\n')
         self.gemspec_mapping.translate(b''.join(parts))
 
     def test_revision_metadata_indexer(self):
         metadata_indexer = RevisionMetadataIndexer(
             config=REVISION_METADATA_CONFIG)
         fill_obj_storage(metadata_indexer.objstorage)
         fill_storage(metadata_indexer.storage)
 
         tool = metadata_indexer.idx_storage.indexer_configuration_get(
             {'tool_'+k: v for (k, v) in TRANSLATOR_TOOL.items()})
         assert tool is not None
 
         metadata_indexer.idx_storage.content_metadata_add([{
             'indexer_configuration_id': tool['id'],
             'id': b'cde',
             'metadata': YARN_PARSER_METADATA,
         }])
 
         sha1_gits = [
             hash_to_bytes('8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
         ]
         metadata_indexer.run(sha1_gits, 'update-dups')
 
         results = list(
             metadata_indexer.idx_storage.
             revision_intrinsic_metadata_get(sha1_gits))
 
         expected_results = [{
             'id': hash_to_bytes('8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
             'tool': TRANSLATOR_TOOL,
             'metadata': YARN_PARSER_METADATA,
             'mappings': ['npm'],
         }]
 
         for result in results:
             del result['tool']['id']
 
         # then
         self.assertEqual(expected_results, results)
 
     def test_revision_metadata_indexer_single_root_dir(self):
         metadata_indexer = RevisionMetadataIndexer(
             config=REVISION_METADATA_CONFIG)
         fill_obj_storage(metadata_indexer.objstorage)
         fill_storage(metadata_indexer.storage)
 
         # Add a parent directory, that is the only directory at the root
         # of the revision
         rev_id = hash_to_bytes('8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f')
         rev = metadata_indexer.storage._revisions[rev_id]
         subdir_id = rev.directory
         rev.directory = b'123456'
         metadata_indexer.storage.directory_add([{
             'id': b'123456',
             'entries': [{
                 'name': b'foobar-1.0.0',
                 'type': 'dir',
                 'target': subdir_id,
                 'perms': 16384,
             }],
         }])
 
         tool = metadata_indexer.idx_storage.indexer_configuration_get(
             {'tool_'+k: v for (k, v) in TRANSLATOR_TOOL.items()})
         assert tool is not None
 
         metadata_indexer.idx_storage.content_metadata_add([{
             'indexer_configuration_id': tool['id'],
             'id': b'cde',
             'metadata': YARN_PARSER_METADATA,
         }])
 
         sha1_gits = [
             hash_to_bytes('8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
         ]
         metadata_indexer.run(sha1_gits, 'update-dups')
 
         results = list(
             metadata_indexer.idx_storage.
             revision_intrinsic_metadata_get(sha1_gits))
 
         expected_results = [{
             'id': hash_to_bytes('8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
             'tool': TRANSLATOR_TOOL,
             'metadata': YARN_PARSER_METADATA,
             'mappings': ['npm'],
         }]
 
         for result in results:
             del result['tool']['id']
 
         # then
         self.assertEqual(expected_results, results)
diff --git a/swh/indexer/tests/test_mimetype.py b/swh/indexer/tests/test_mimetype.py
index 72a0503..cf39bc3 100644
--- a/swh/indexer/tests/test_mimetype.py
+++ b/swh/indexer/tests/test_mimetype.py
@@ -1,145 +1,147 @@
 # Copyright (C) 2017-2018  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import pytest
 import unittest
 
+from typing import Any, Dict
+
 from swh.indexer.mimetype import (
     MimetypeIndexer, MimetypeRangeIndexer, compute_mimetype_encoding
 )
 
 from swh.indexer.tests.utils import (
     CommonContentIndexerTest, CommonContentIndexerRangeTest,
     BASE_TEST_CONFIG, fill_storage, fill_obj_storage, filter_dict,
 )
 
 
 class BasicTest(unittest.TestCase):
     def test_compute_mimetype_encoding(self):
         """Compute mimetype encoding should return results"""
         for _input, _mimetype, _encoding in [
                 ('du français'.encode(), 'text/plain', 'utf-8'),
                 (b'def __init__(self):', 'text/x-python', 'us-ascii')]:
 
             actual_result = compute_mimetype_encoding(_input)
             self.assertEqual(actual_result, {
                 'mimetype': _mimetype,
                 'encoding': _encoding
             })
 
 
 CONFIG = {
     **BASE_TEST_CONFIG,
     'tools': {
         'name': 'file',
         'version': '1:5.30-1+deb9u1',
         'configuration': {
             "type": "library",
             "debian-package": "python3-magic"
         },
     },
-}
+}  # type: Dict[str, Any]
 
 
 class TestMimetypeIndexer(CommonContentIndexerTest, unittest.TestCase):
     """Mimetype indexer test scenarios:
 
     - Known sha1s in the input list have their data indexed
     - Unknown sha1 in the input list are not indexed
 
     """
     legacy_get_format = True
 
     def get_indexer_results(self, ids):
         yield from self.idx_storage.content_mimetype_get(ids)
 
     def setUp(self):
         self.indexer = MimetypeIndexer(config=CONFIG)
         self.indexer.catch_exceptions = False
         self.idx_storage = self.indexer.idx_storage
         fill_storage(self.indexer.storage)
         fill_obj_storage(self.indexer.objstorage)
 
         self.id0 = '01c9379dfc33803963d07c1ccc748d3fe4c96bb5'
         self.id1 = '688a5ef812c53907562fe379d4b3851e69c7cb15'
         self.id2 = 'da39a3ee5e6b4b0d3255bfef95601890afd80709'
 
         tool = {k.replace('tool_', ''): v
                 for (k, v) in self.indexer.tool.items()}
 
         self.expected_results = {
             self.id0: {
                 'id': self.id0,
                 'tool': tool,
                 'mimetype': 'text/plain',
                 'encoding': 'us-ascii',
             },
             self.id1: {
                 'id': self.id1,
                 'tool': tool,
                 'mimetype': 'text/plain',
                 'encoding': 'us-ascii',
             },
             self.id2: {
                 'id': self.id2,
                 'tool': tool,
                 'mimetype': 'application/x-empty',
                 'encoding': 'binary',
             }
         }
 
 
 RANGE_CONFIG = dict(list(CONFIG.items()) + [('write_batch_size', 100)])
 
 
 class TestMimetypeRangeIndexer(
         CommonContentIndexerRangeTest, unittest.TestCase):
     """Range Mimetype Indexer tests.
 
     - new data within range are indexed
     - no data outside a range are indexed
     - with filtering existing indexed data prior to compute new index
     - without filtering existing indexed data prior to compute new index
 
     """
     def setUp(self):
         super().setUp()
         self.indexer = MimetypeRangeIndexer(config=RANGE_CONFIG)
         self.indexer.catch_exceptions = False
         fill_storage(self.indexer.storage)
         fill_obj_storage(self.indexer.objstorage)
 
         self.id0 = '01c9379dfc33803963d07c1ccc748d3fe4c96bb5'
         self.id1 = '02fb2c89e14f7fab46701478c83779c7beb7b069'
         self.id2 = '103bc087db1d26afc3a0283f38663d081e9b01e6'
         tool_id = self.indexer.tool['id']
 
         self.expected_results = {
             self.id0: {
                 'encoding': 'us-ascii',
                 'id': self.id0,
                 'indexer_configuration_id': tool_id,
                 'mimetype': 'text/plain'},
             self.id1: {
                 'encoding': 'us-ascii',
                 'id': self.id1,
                 'indexer_configuration_id': tool_id,
                 'mimetype': 'text/x-python'},
             self.id2: {
                 'encoding': 'us-ascii',
                 'id': self.id2,
                 'indexer_configuration_id': tool_id,
                 'mimetype': 'text/plain'}
         }
 
 
 def test_mimetype_w_no_tool():
     with pytest.raises(ValueError):
         MimetypeIndexer(config=filter_dict(CONFIG, 'tools'))
 
 
 def test_mimetype_range_w_no_tool():
     with pytest.raises(ValueError):
         MimetypeRangeIndexer(config=filter_dict(CONFIG, 'tools'))
diff --git a/swh/indexer/tests/utils.py b/swh/indexer/tests/utils.py
index f09927e..6b5860b 100644
--- a/swh/indexer/tests/utils.py
+++ b/swh/indexer/tests/utils.py
@@ -1,749 +1,755 @@
 # Copyright (C) 2017-2019  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import abc
 import datetime
 import functools
 import random
 import unittest
 
 from hypothesis import strategies
 
 from swh.model import hashutil
 from swh.model.hashutil import hash_to_bytes, hash_to_hex
 
 from swh.indexer.storage import INDEXER_CFG_KEY
 
 BASE_TEST_CONFIG = {
     'storage': {
         'cls': 'memory',
         'args': {
         },
     },
     'objstorage': {
         'cls': 'memory',
         'args': {
         },
     },
     INDEXER_CFG_KEY: {
         'cls': 'memory',
         'args': {
         },
     },
 }
 
 ORIGINS = [
         {
             'type': 'git',
             'url': 'https://github.com/SoftwareHeritage/swh-storage'},
         {
             'type': 'ftp',
             'url': 'rsync://ftp.gnu.org/gnu/3dldf'},
         {
             'type': 'deposit',
             'url': 'https://forge.softwareheritage.org/source/jesuisgpl/'},
         {
             'type': 'pypi',
             'url': 'https://pypi.org/project/limnoria/'},
         {
             'type': 'svn',
             'url': 'http://0-512-md.googlecode.com/svn/'},
         {
             'type': 'git',
             'url': 'https://github.com/librariesio/yarn-parser'},
         {
             'type': 'git',
             'url': 'https://github.com/librariesio/yarn-parser.git'},
         ]
 
 SNAPSHOTS = [
     {
         'origin': 'https://github.com/SoftwareHeritage/swh-storage',
         'branches': {
             b'refs/heads/add-revision-origin-cache': {
                 'target': b'L[\xce\x1c\x88\x8eF\t\xf1"\x19\x1e\xfb\xc0'
                           b's\xe7/\xe9l\x1e',
                 'target_type': 'revision'},
             b'HEAD': {
                 'target': b'8K\x12\x00d\x03\xcc\xe4]bS\xe3\x8f{\xd7}'
                           b'\xac\xefrm',
                 'target_type': 'revision'},
             b'refs/tags/v0.0.103': {
                 'target': b'\xb6"Im{\xfdLb\xb0\x94N\xea\x96m\x13x\x88+'
                           b'\x0f\xdd',
                 'target_type': 'release'},
             }},
     {
         'origin': 'rsync://ftp.gnu.org/gnu/3dldf',
         'branches': {
             b'3DLDF-1.1.4.tar.gz': {
                 'target': b'dJ\xfb\x1c\x91\xf4\x82B%]6\xa2\x90|\xd3\xfc'
                           b'"G\x99\x11',
                 'target_type': 'revision'},
             b'3DLDF-2.0.2.tar.gz': {
                 'target': b'\xb6\x0e\xe7\x9e9\xac\xaa\x19\x9e='
                           b'\xd1\xc5\x00\\\xc6\xfc\xe0\xa6\xb4V',
                 'target_type': 'revision'},
             b'3DLDF-2.0.3-examples.tar.gz': {
                 'target': b'!H\x19\xc0\xee\x82-\x12F1\xbd\x97'
                           b'\xfe\xadZ\x80\x80\xc1\x83\xff',
                 'target_type': 'revision'},
             b'3DLDF-2.0.3.tar.gz': {
                 'target': b'\x8e\xa9\x8e/\xea}\x9feF\xf4\x9f\xfd\xee'
                           b'\xcc\x1a\xb4`\x8c\x8by',
                 'target_type': 'revision'},
             b'3DLDF-2.0.tar.gz': {
                 'target': b'F6*\xff(?\x19a\xef\xb6\xc2\x1fv$S\xe3G'
                           b'\xd3\xd1m',
                 'target_type': 'revision'}
             }},
     {
         'origin': 'https://forge.softwareheritage.org/source/jesuisgpl/',
         'branches': {
             b'master': {
                 'target': b'\xe7n\xa4\x9c\x9f\xfb\xb7\xf76\x11\x08{'
                           b'\xa6\xe9\x99\xb1\x9e]q\xeb',
                 'target_type': 'revision'}
         },
         'id': b"h\xc0\xd2a\x04\xd4~'\x8d\xd6\xbe\x07\xeda\xfa\xfbV"
               b"\x1d\r "},
     {
         'origin': 'https://pypi.org/project/limnoria/',
         'branches': {
             b'HEAD': {
                 'target': b'releases/2018.09.09',
                 'target_type': 'alias'},
             b'releases/2018.09.01': {
                 'target': b'<\xee1(\xe8\x8d_\xc1\xc9\xa6rT\xf1\x1d'
                           b'\xbb\xdfF\xfdw\xcf',
                 'target_type': 'revision'},
             b'releases/2018.09.09': {
                 'target': b'\x83\xb9\xb6\xc7\x05\xb1%\xd0\xfem\xd8k'
                           b'A\x10\x9d\xc5\xfa2\xf8t',
                 'target_type': 'revision'}},
         'id': b'{\xda\x8e\x84\x7fX\xff\x92\x80^\x93V\x18\xa3\xfay'
               b'\x12\x9e\xd6\xb3'},
     {
         'origin': 'http://0-512-md.googlecode.com/svn/',
         'branches': {
             b'master': {
                 'target': b'\xe4?r\xe1,\x88\xab\xec\xe7\x9a\x87\xb8'
                           b'\xc9\xad#.\x1bw=\x18',
                 'target_type': 'revision'}},
         'id': b'\xa1\xa2\x8c\n\xb3\x87\xa8\xf9\xe0a\x8c\xb7'
               b'\x05\xea\xb8\x1f\xc4H\xf4s'},
     {
         'origin': 'https://github.com/librariesio/yarn-parser',
         'branches': {
             b'HEAD': {
                 'target': hash_to_bytes(
                     '8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
                 'target_type': 'revision'}}},
     {
         'origin': 'https://github.com/librariesio/yarn-parser.git',
         'branches': {
             b'HEAD': {
                 'target': hash_to_bytes(
                     '8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
                 'target_type': 'revision'}}},
 ]
 
 
 REVISIONS = [{
     'id': hash_to_bytes('8dbb6aeb036e7fd80664eb8bfd1507881af1ba9f'),
     'message': 'Improve search functionality',
     'author': {
         'name': b'Andrew Nesbitt',
         'fullname': b'Andrew Nesbitt <andrewnez@gmail.com>',
         'email': b'andrewnez@gmail.com'
     },
     'committer': {
         'name': b'Andrew Nesbitt',
         'fullname': b'Andrew Nesbitt <andrewnez@gmail.com>',
         'email': b'andrewnez@gmail.com'
     },
     'committer_date': {
         'negative_utc': None,
         'offset': 120,
         'timestamp': {
             'microseconds': 0,
             'seconds': 1380883849
         }
     },
     'type': 'git',
     'synthetic': False,
     'date': {
         'negative_utc': False,
         'timestamp': {
             'seconds': 1487596456,
             'microseconds': 0
         },
         'offset': 0
     },
     'directory': b'10'
 }]
 
 DIRECTORY_ID = b'10'
 
 DIRECTORY_ENTRIES = [{
     'name': b'index.js',
     'type': 'file',
     'target': b'abc',
     'perms': 33188,
     },
     {
     'name': b'package.json',
     'type': 'file',
     'target': b'cde',
     'perms': 33188,
     },
     {
     'name': b'.github',
     'type': 'dir',
     'target': b'11',
     'perms': 16384,
     }
 ]
 
 SHA1_TO_LICENSES = {
     '01c9379dfc33803963d07c1ccc748d3fe4c96bb5': ['GPL'],
     '02fb2c89e14f7fab46701478c83779c7beb7b069': ['Apache2.0'],
     '103bc087db1d26afc3a0283f38663d081e9b01e6': ['MIT'],
     '688a5ef812c53907562fe379d4b3851e69c7cb15': ['AGPL'],
     'da39a3ee5e6b4b0d3255bfef95601890afd80709': [],
 }
 
 
 SHA1_TO_CTAGS = {
     '01c9379dfc33803963d07c1ccc748d3fe4c96bb5': [{
         'name': 'foo',
         'kind': 'str',
         'line': 10,
         'lang': 'bar',
     }],
     'd4c647f0fc257591cc9ba1722484229780d1c607': [{
         'name': 'let',
         'kind': 'int',
         'line': 100,
         'lang': 'haskell',
     }],
     '688a5ef812c53907562fe379d4b3851e69c7cb15': [{
         'name': 'symbol',
         'kind': 'float',
         'line': 99,
         'lang': 'python',
     }],
 }
 
 
 OBJ_STORAGE_DATA = {
     '01c9379dfc33803963d07c1ccc748d3fe4c96bb5': b'this is some text',
     '688a5ef812c53907562fe379d4b3851e69c7cb15': b'another text',
     '8986af901dd2043044ce8f0d8fc039153641cf17': b'yet another text',
     '02fb2c89e14f7fab46701478c83779c7beb7b069': b"""
     import unittest
     import logging
     from swh.indexer.mimetype import MimetypeIndexer
     from swh.indexer.tests.test_utils import MockObjStorage
 
     class MockStorage():
         def content_mimetype_add(self, mimetypes):
             self.state = mimetypes
             self.conflict_update = conflict_update
 
         def indexer_configuration_add(self, tools):
             return [{
                 'id': 10,
             }]
     """,
     '103bc087db1d26afc3a0283f38663d081e9b01e6': b"""
         #ifndef __AVL__
         #define __AVL__
 
         typedef struct _avl_tree avl_tree;
 
         typedef struct _data_t {
           int content;
         } data_t;
     """,
     '93666f74f1cf635c8c8ac118879da6ec5623c410': b"""
     (should 'pygments (recognize 'lisp 'easily))
 
     """,
     '26a9f72a7c87cc9205725cfd879f514ff4f3d8d5': b"""
     {
         "name": "test_metadata",
         "version": "0.0.1",
         "description": "Simple package.json test for indexer",
         "repository": {
           "type": "git",
           "url": "https://github.com/moranegg/metadata_test"
       }
     }
     """,
     'd4c647f0fc257591cc9ba1722484229780d1c607': b"""
     {
       "version": "5.0.3",
       "name": "npm",
       "description": "a package manager for JavaScript",
       "keywords": [
         "install",
         "modules",
         "package manager",
         "package.json"
       ],
       "preferGlobal": true,
       "config": {
         "publishtest": false
       },
       "homepage": "https://docs.npmjs.com/",
       "author": "Isaac Z. Schlueter <i@izs.me> (http://blog.izs.me)",
       "repository": {
         "type": "git",
         "url": "https://github.com/npm/npm"
       },
       "bugs": {
         "url": "https://github.com/npm/npm/issues"
       },
       "dependencies": {
         "JSONStream": "~1.3.1",
         "abbrev": "~1.1.0",
         "ansi-regex": "~2.1.1",
         "ansicolors": "~0.3.2",
         "ansistyles": "~0.1.3"
       },
       "devDependencies": {
         "tacks": "~1.2.6",
         "tap": "~10.3.2"
       },
       "license": "Artistic-2.0"
     }
 
     """,
     'a7ab314d8a11d2c93e3dcf528ca294e7b431c449': b"""
     """,
     'da39a3ee5e6b4b0d3255bfef95601890afd80709': b'',
     # 626364
     hash_to_hex(b'bcd'): b'unimportant content for bcd',
     # 636465
     hash_to_hex(b'cde'): b"""
     {
       "name": "yarn-parser",
       "version": "1.0.0",
       "description": "Tiny web service for parsing yarn.lock files",
       "main": "index.js",
       "scripts": {
         "start": "node index.js",
         "test": "mocha"
       },
       "engines": {
         "node": "9.8.0"
       },
       "repository": {
         "type": "git",
         "url": "git+https://github.com/librariesio/yarn-parser.git"
       },
       "keywords": [
         "yarn",
         "parse",
         "lock",
         "dependencies"
       ],
       "author": "Andrew Nesbitt",
       "license": "AGPL-3.0",
       "bugs": {
         "url": "https://github.com/librariesio/yarn-parser/issues"
       },
       "homepage": "https://github.com/librariesio/yarn-parser#readme",
       "dependencies": {
         "@yarnpkg/lockfile": "^1.0.0",
         "body-parser": "^1.15.2",
         "express": "^4.14.0"
       },
       "devDependencies": {
         "chai": "^4.1.2",
         "mocha": "^5.2.0",
         "request": "^2.87.0",
         "test": "^0.6.0"
       }
     }
 
 """
 }
 
 YARN_PARSER_METADATA = {
     '@context': 'https://doi.org/10.5063/schema/codemeta-2.0',
     'url':
         'https://github.com/librariesio/yarn-parser#readme',
     'codeRepository':
         'git+git+https://github.com/librariesio/yarn-parser.git',
     'author': [{
         'type': 'Person',
         'name': 'Andrew Nesbitt'
     }],
     'license': 'https://spdx.org/licenses/AGPL-3.0',
     'version': '1.0.0',
     'description':
         'Tiny web service for parsing yarn.lock files',
     'issueTracker':
         'https://github.com/librariesio/yarn-parser/issues',
     'name': 'yarn-parser',
     'keywords': ['yarn', 'parse', 'lock', 'dependencies'],
     'type': 'SoftwareSourceCode',
 }
 
 
 json_dict_keys = strategies.one_of(
     strategies.characters(),
-    *map(strategies.just, ['type', 'url', 'name', 'email', '@id',
-                           '@context', 'repository', 'license',
-                           'repositories', 'licenses'
-                           ]),
+    strategies.just('type'),
+    strategies.just('url'),
+    strategies.just('name'),
+    strategies.just('email'),
+    strategies.just('@id'),
+    strategies.just('@context'),
+    strategies.just('repository'),
+    strategies.just('license'),
+    strategies.just('repositories'),
+    strategies.just('licenses'),
 )
 """Hypothesis strategy that generates strings, with an emphasis on those
 that are often used as dictionary keys in metadata files."""
 
 
 generic_json_document = strategies.recursive(
     strategies.none() | strategies.booleans() | strategies.floats() |
     strategies.characters(),
     lambda children: (
         strategies.lists(children, 1) |
         strategies.dictionaries(json_dict_keys, children, min_size=1)
     )
 )
 """Hypothesis strategy that generates possible values for values of JSON
 metadata files."""
 
 
 def json_document_strategy(keys=None):
     """Generates an hypothesis strategy that generates metadata files
     for a JSON-based format that uses the given keys."""
     if keys is None:
         keys = strategies.characters()
     else:
         keys = strategies.one_of(map(strategies.just, keys))
 
     return strategies.dictionaries(keys, generic_json_document, min_size=1)
 
 
 def _tree_to_xml(root, xmlns, data):
     def encode(s):
         "Skips unpaired surrogates generated by json_document_strategy"
         return s.encode('utf8', 'replace')
 
     def to_xml(data, indent=b' '):
         if data is None:
             return b''
         elif isinstance(data, (bool, str, int, float)):
             return indent + encode(str(data))
         elif isinstance(data, list):
             return b'\n'.join(to_xml(v, indent=indent) for v in data)
         elif isinstance(data, dict):
             lines = []
             for (key, value) in data.items():
                 lines.append(indent + encode('<{}>'.format(key)))
                 lines.append(to_xml(value, indent=indent+b' '))
                 lines.append(indent + encode('</{}>'.format(key)))
             return b'\n'.join(lines)
         else:
             raise TypeError(data)
 
     return b'\n'.join([
         '<{} xmlns="{}">'.format(root, xmlns).encode(),
         to_xml(data),
         '</{}>'.format(root).encode(),
     ])
 
 
 class TreeToXmlTest(unittest.TestCase):
     def test_leaves(self):
         self.assertEqual(
             _tree_to_xml('root', 'http://example.com', None),
             b'<root xmlns="http://example.com">\n\n</root>'
         )
         self.assertEqual(
             _tree_to_xml('root', 'http://example.com', True),
             b'<root xmlns="http://example.com">\n True\n</root>'
         )
         self.assertEqual(
             _tree_to_xml('root', 'http://example.com', 'abc'),
             b'<root xmlns="http://example.com">\n abc\n</root>'
         )
         self.assertEqual(
             _tree_to_xml('root', 'http://example.com', 42),
             b'<root xmlns="http://example.com">\n 42\n</root>'
         )
         self.assertEqual(
             _tree_to_xml('root', 'http://example.com', 3.14),
             b'<root xmlns="http://example.com">\n 3.14\n</root>'
         )
 
     def test_dict(self):
         self.assertIn(
             _tree_to_xml('root', 'http://example.com', {
                 'foo': 'bar',
                 'baz': 'qux'
             }),
             [
                 b'<root xmlns="http://example.com">\n'
                 b' <foo>\n  bar\n </foo>\n'
                 b' <baz>\n  qux\n </baz>\n'
                 b'</root>',
                 b'<root xmlns="http://example.com">\n'
                 b' <baz>\n  qux\n </baz>\n'
                 b' <foo>\n  bar\n </foo>\n'
                 b'</root>'
             ]
         )
 
     def test_list(self):
         self.assertEqual(
             _tree_to_xml('root', 'http://example.com', [
                 {'foo': 'bar'},
                 {'foo': 'baz'},
             ]),
             b'<root xmlns="http://example.com">\n'
             b' <foo>\n  bar\n </foo>\n'
             b' <foo>\n  baz\n </foo>\n'
             b'</root>'
         )
 
 
 def xml_document_strategy(keys, root, xmlns):
     """Generates an hypothesis strategy that generates metadata files
     for an XML format that uses the given keys."""
 
     return strategies.builds(
         functools.partial(_tree_to_xml, root, xmlns),
         json_document_strategy(keys))
 
 
 def filter_dict(d, keys):
     'return a copy of the dict with keys deleted'
     if not isinstance(keys, (list, tuple)):
         keys = (keys, )
     return dict((k, v) for (k, v) in d.items() if k not in keys)
 
 
 def fill_obj_storage(obj_storage):
     """Add some content in an object storage."""
     for (obj_id, content) in OBJ_STORAGE_DATA.items():
         obj_storage.add(content, obj_id=hash_to_bytes(obj_id))
 
 
 def fill_storage(storage):
     for origin in ORIGINS:
         storage.origin_add_one(origin)
     for snap in SNAPSHOTS:
         origin_url = snap['origin']
         visit = storage.origin_visit_add(origin_url, datetime.datetime.now())
         snap_id = snap.get('id') or \
             bytes([random.randint(0, 255) for _ in range(32)])
         storage.snapshot_add([{
             'id': snap_id,
             'branches': snap['branches']
         }])
         storage.origin_visit_update(
             origin_url, visit['visit'], status='full', snapshot=snap_id)
     storage.revision_add(REVISIONS)
 
     contents = []
     for (obj_id, content) in OBJ_STORAGE_DATA.items():
         content_hashes = hashutil.MultiHash.from_data(content).digest()
         contents.append({
             'data': content,
             'length': len(content),
             'status': 'visible',
             'sha1': hash_to_bytes(obj_id),
             'sha1_git': hash_to_bytes(obj_id),
             'sha256': content_hashes['sha256'],
             'blake2s256': content_hashes['blake2s256']
         })
     storage.content_add(contents)
     storage.directory_add([{
         'id': DIRECTORY_ID,
         'entries': DIRECTORY_ENTRIES,
     }])
 
 
 class CommonContentIndexerTest(metaclass=abc.ABCMeta):
     legacy_get_format = False
     """True if and only if the tested indexer uses the legacy format.
     see: https://forge.softwareheritage.org/T1433
 
     """
     def get_indexer_results(self, ids):
         """Override this for indexers that don't have a mock storage."""
         return self.indexer.idx_storage.state
 
     def assert_legacy_results_ok(self, sha1s, expected_results=None):
         # XXX old format, remove this when all endpoints are
         #     updated to the new one
         #     see: https://forge.softwareheritage.org/T1433
         sha1s = [sha1 if isinstance(sha1, bytes) else hash_to_bytes(sha1)
                  for sha1 in sha1s]
         actual_results = list(self.get_indexer_results(sha1s))
 
         if expected_results is None:
             expected_results = self.expected_results
 
         self.assertEqual(len(expected_results), len(actual_results),
                          (expected_results, actual_results))
         for indexed_data in actual_results:
             _id = indexed_data['id']
             expected_data = expected_results[hashutil.hash_to_hex(_id)].copy()
             expected_data['id'] = _id
             self.assertEqual(indexed_data, expected_data)
 
     def assert_results_ok(self, sha1s, expected_results=None):
         if self.legacy_get_format:
             self.assert_legacy_results_ok(sha1s, expected_results)
             return
 
         sha1s = [sha1 if isinstance(sha1, bytes) else hash_to_bytes(sha1)
                  for sha1 in sha1s]
         actual_results = list(self.get_indexer_results(sha1s))
 
         if expected_results is None:
             expected_results = self.expected_results
 
         self.assertEqual(len(expected_results), len(actual_results),
                          (expected_results, actual_results))
         for indexed_data in actual_results:
             (_id, indexed_data) = list(indexed_data.items())[0]
             expected_data = expected_results[hashutil.hash_to_hex(_id)].copy()
             expected_data = [expected_data]
             self.assertEqual(indexed_data, expected_data)
 
     def test_index(self):
         """Known sha1 have their data indexed
 
         """
         sha1s = [self.id0, self.id1, self.id2]
 
         # when
         self.indexer.run(sha1s, policy_update='update-dups')
 
         self.assert_results_ok(sha1s)
 
         # 2nd pass
         self.indexer.run(sha1s, policy_update='ignore-dups')
 
         self.assert_results_ok(sha1s)
 
     def test_index_one_unknown_sha1(self):
         """Unknown sha1 are not indexed"""
         sha1s = [self.id1,
                  '799a5ef812c53907562fe379d4b3851e69c7cb15',  # unknown
                  '800a5ef812c53907562fe379d4b3851e69c7cb15']  # unknown
 
         # when
         self.indexer.run(sha1s, policy_update='update-dups')
 
         # then
         expected_results = {
             k: v for k, v in self.expected_results.items() if k in sha1s
         }
 
         self.assert_results_ok(sha1s, expected_results)
 
 
 class CommonContentIndexerRangeTest:
     """Allows to factorize tests on range indexer.
 
     """
     def setUp(self):
         self.contents = sorted(OBJ_STORAGE_DATA)
 
     def assert_results_ok(self, start, end, actual_results,
                           expected_results=None):
         if expected_results is None:
             expected_results = self.expected_results
 
         actual_results = list(actual_results)
         for indexed_data in actual_results:
             _id = indexed_data['id']
             assert isinstance(_id, bytes)
             indexed_data = indexed_data.copy()
             indexed_data['id'] = hash_to_hex(indexed_data['id'])
             self.assertEqual(indexed_data, expected_results[hash_to_hex(_id)])
             self.assertTrue(start <= _id <= end)
             _tool_id = indexed_data['indexer_configuration_id']
             self.assertEqual(_tool_id, self.indexer.tool['id'])
 
     def test__index_contents(self):
         """Indexing contents without existing data results in indexed data
 
         """
         _start, _end = [self.contents[0], self.contents[2]]  # output hex ids
         start, end = map(hashutil.hash_to_bytes, (_start, _end))
         # given
         actual_results = list(self.indexer._index_contents(
             start, end, indexed={}))
 
         self.assert_results_ok(start, end, actual_results)
 
     def test__index_contents_with_indexed_data(self):
         """Indexing contents with existing data results in less indexed data
 
         """
         _start, _end = [self.contents[0], self.contents[2]]  # output hex ids
         start, end = map(hashutil.hash_to_bytes, (_start, _end))
         data_indexed = [self.id0, self.id2]
 
         # given
         actual_results = self.indexer._index_contents(
             start, end, indexed=set(map(hash_to_bytes, data_indexed)))
 
         # craft the expected results
         expected_results = self.expected_results.copy()
         for already_indexed_key in data_indexed:
             expected_results.pop(already_indexed_key)
 
         self.assert_results_ok(
             start, end, actual_results, expected_results)
 
     def test_generate_content_get(self):
         """Optimal indexing should result in indexed data
 
         """
         _start, _end = [self.contents[0], self.contents[2]]  # output hex ids
         start, end = map(hashutil.hash_to_bytes, (_start, _end))
 
         # given
         actual_results = self.indexer.run(start, end)
 
         # then
         self.assertTrue(actual_results)
 
     def test_generate_content_get_input_as_bytes(self):
         """Optimal indexing should result in indexed data
 
         Input are in bytes here.
 
         """
         _start, _end = [self.contents[0], self.contents[2]]  # output hex ids
         start, end = map(hashutil.hash_to_bytes, (_start, _end))
 
         # given
         actual_results = self.indexer.run(  # checks the bytes input this time
             start, end, skip_existing=False)
         # no already indexed data so same result as prior test
 
         # then
         self.assertTrue(actual_results)
 
     def test_generate_content_get_no_result(self):
         """No result indexed returns False"""
         _start, _end = ['0000000000000000000000000000000000000000',
                         '0000000000000000000000000000000000000001']
         start, end = map(hashutil.hash_to_bytes, (_start, _end))
         # given
         actual_results = self.indexer.run(
             start, end, incremental=False)
 
         # then
         self.assertFalse(actual_results)
diff --git a/tox.ini b/tox.ini
index fb80d5b..fca1b6b 100644
--- a/tox.ini
+++ b/tox.ini
@@ -1,41 +1,49 @@
 [tox]
-envlist=flake8,py3
+envlist=flake8,mypy,py3
 
 [testenv:py3]
 deps =
   .[testing]
   pytest-cov
   pifpaf
 commands =
   pifpaf run postgresql -- pytest --doctest-modules --hypothesis-profile=fast \
          {envsitepackagesdir}/swh/indexer \
          --cov={envsitepackagesdir}/swh/indexer \
          --cov-branch {posargs}
 
 [testenv:py3-slow]
 deps =
   .[testing]
   pytest-cov
   pifpaf
 commands =
   pifpaf run postgresql -- pytest --doctest-modules --hypothesis-profile=slow \
          {envsitepackagesdir}/swh/indexer \
          --cov={envsitepackagesdir}/swh/indexer \
          --cov-branch {posargs}
 
 [testenv:py3-prop]
 deps =
   .[testing]
   pytest-cov
   pifpaf
 commands =
   pifpaf run postgresql -- pytest --doctest-modules --hypothesis-profile=fast \
          -m property_based --disable-warnings \
          {envsitepackagesdir}/swh/indexer
 
 [testenv:flake8]
 skip_install = true
 deps =
   flake8
 commands =
   {envpython} -m flake8
+
+[testenv:mypy]
+skip_install = true
+deps =
+  .[testing]
+  mypy
+commands =
+  mypy swh