diff --git a/PKG-INFO b/PKG-INFO
index 7c5cbd3..4912824 100644
--- a/PKG-INFO
+++ b/PKG-INFO
@@ -1,71 +1,71 @@
 Metadata-Version: 2.1
 Name: swh.indexer
-Version: 2.8.0
+Version: 2.9.0
 Summary: Software Heritage Content Indexer
 Home-page: https://forge.softwareheritage.org/diffusion/78/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-indexer
 Project-URL: Documentation, https://docs.softwareheritage.org/devel/swh-indexer/
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Requires-Python: >=3.7
 Description-Content-Type: text/markdown
 Provides-Extra: testing
 License-File: LICENSE
 License-File: AUTHORS
 
 swh-indexer
 ============
 
 Tools to compute multiple indexes on SWH's raw contents:
 - content:
   - mimetype
   - ctags
   - language
   - fossology-license
   - metadata
 - revision:
   - metadata
 
 An indexer is in charge of:
 - looking up objects
 - extracting information from those objects
 - store those information in the swh-indexer db
 
 There are multiple indexers working on different object types:
   - content indexer: works with content sha1 hashes
   - revision indexer: works with revision sha1 hashes
   - origin indexer: works with origin identifiers
 
 Indexation procedure:
 - receive batch of ids
 - retrieve the associated data depending on object type
 - compute for that object some index
 - store the result to swh's storage
 
 Current content indexers:
 
 - mimetype (queue swh_indexer_content_mimetype): detect the encoding
   and mimetype
 
 - language (queue swh_indexer_content_language): detect the
   programming language
 
 - ctags (queue swh_indexer_content_ctags): compute tags information
 
 - fossology-license (queue swh_indexer_fossology_license): compute the
   license
 
 - metadata: translate file into translated_metadata dict
 
 Current revision indexers:
 
 - metadata: detects files containing metadata and retrieves translated_metadata
   in content_metadata table in storage or run content indexer to translate
   files.
diff --git a/debian/changelog b/debian/changelog
index b4a4a0e..c77055f 100644
--- a/debian/changelog
+++ b/debian/changelog
@@ -1,1610 +1,1614 @@
-swh-indexer (2.8.0-1~swh1~bpo10+1) buster-swh; urgency=medium
+swh-indexer (2.9.0-1~swh1) unstable-swh; urgency=medium
 
-  * Rebuild for buster-swh
+  * New upstream release 2.9.0     - (tagged by Valentin Lorentz
+    <vlorentz@softwareheritage.org> on 2022-11-29 15:28:28 +0100)
+  * Upstream changes:     - v2.9.0     - * storage: Insert from
+    temporary tables in consistent order     - * Drop content_language
+    and content_ctags tables and related SQL functions
 
- -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 23 Nov 2022 08:09:06 +0000
+ -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 29 Nov 2022 14:35:17 +0000
 
 swh-indexer (2.8.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.8.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-11-23 08:57:20 +0100)
   * Upstream changes:     - v2.8.0     - * journal writer: only flush
     kafka once per batch     - * origin_head: Do not fetch complete
     snapshots for non-FTP visits     - * ExtrinsicMetadataIndexer: Add
     support for metadata with origin in context
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 23 Nov 2022 08:04:03 +0000
 
 swh-indexer (2.7.3-2~swh1) unstable-swh; urgency=medium
 
   * Fix debian package build by adding debian/pybuild.testfiles
 
  -- Antoine Lambert <anlambert@softwareheritage.org>  Wed, 02 Nov 2022 19:34:51 +0100
 
 swh-indexer (2.7.3-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.7.3     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-11-02 17:42:38 +0100)
   * Upstream changes:     - v2.7.3     - * codemeta: Fix crash on SWORD
     documents that specify an id
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 02 Nov 2022 16:48:00 +0000
 
 swh-indexer (2.7.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.7.2     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-10-27 14:24:43 +0200)
   * Upstream changes:     - v2.7.2     - * Reset Sentry tags when
     leaving an object's context     - * metadata: Make default tool
     configuration follow swh.indexer versions     - * Fix crashes in
     translation (mostly from NPM and Maven)     - * Fix incorrect
     outputs when translating from SWORD
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 27 Oct 2022 12:35:51 +0000
 
 swh-indexer (2.7.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.7.1     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-10-07 12:01:09 +0200)
   * Upstream changes:     - v2.7.1     - * npm: Fix crash on invalid
     URLs in 'bugs' field.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 07 Oct 2022 10:10:37 +0000
 
 swh-indexer (2.6.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.6.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-09-12 10:55:11 +0200)
   * Upstream changes:     - v2.6.0     - * Convert SWHID to str before
     passing to sentry_sdk.set_tag     - * Fix various crashes     - *
     github: Add support for 'topics'     - * npm, maven: ignore
     blatantly invalid licenses and URLs     - * cli: Pass all
     journal_client config keys to the JournalClient
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 12 Sep 2022 09:07:01 +0000
 
 swh-indexer (2.5.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.5.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-08-31 18:10:38
     +0200)
   * Upstream changes:     - v2.5.0     - indexer.cli: Allow batch_size
     configuration on journal client
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 31 Aug 2022 16:20:18 +0000
 
 swh-indexer (2.4.4-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.4.4     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-08-31 11:26:51 +0200)
   * Upstream changes:     - v2.4.4     - * Revert "metadata: Drop
     unsupported key 'type'"     - * rehash: Call
     objstorage.content_get() with a HashDict instead of single hash
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 31 Aug 2022 09:36:21 +0000
 
 swh-indexer (2.4.3-1~swh2) unstable-swh; urgency=medium
 
   * Drop blocking dependency constraint and bump new version.
 
  -- Antoine R. Dumont (@ardumont) <ardumont@softwareheritage.org>  Tue, 30 Aug 2022 15:37:01 +0200
 
 swh-indexer (2.4.3-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.4.3     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-08-30 11:09:04
     +0200)
   * Upstream changes:     - v2.4.3     - metadata: Drop unsupported key
     'type'
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 30 Aug 2022 09:25:28 +0000
 
 swh-indexer (2.4.2-1~swh2) unstable-swh; urgency=medium
 
   * Bump new release
 
  -- Antoine R. Dumont (@ardumont) <ardumont@softwareheritage.org>  Thu, 25 Aug 2022 14:30:54 +0200
 
 swh-indexer (2.4.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.4.2     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-08-25 13:24:10 +0200)
   * Upstream changes:     - v2.4.2     - * Re-trigger Debian build
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 25 Aug 2022 11:33:19 +0000
 
 swh-indexer (2.4.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.4.1     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-08-25 12:22:48 +0200)
   * Upstream changes:     - v2.4.1     - * metadata_dictionary: Fix
     crash on null list item in an uri_field.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 25 Aug 2022 10:32:04 +0000
 
 swh-indexer (2.4.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.4.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-08-25 11:58:05 +0200)
   * Upstream changes:     - v2.4.0     - * metadata_dictionary: Add
     mappings for "*.nuspec" files     - * Refactor metadata mappings
     using rdflib.Graph instead of JSON-LD internally     - * Other
     internal refactorings     - * metadata_dictionary: Add mapping for
     SWORD/Atom with Codemeta
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 25 Aug 2022 10:09:34 +0000
 
 swh-indexer (2.3.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.3.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-08-10 12:16:48 +0200)
   * Upstream changes:     - v2.3.0     - * Tag Sentry events with object
     ids     - * Fix crashes on incorrect URLs in `@id`     - * Fix crash
     on null characters in JSON     - * Fix support of old
     RawExtrinsicMetadata objects with no id
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 10 Aug 2022 10:26:27 +0000
 
 swh-indexer (2.2.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.2.2     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-07-29 13:41:43
     +0200)
   * Upstream changes:     - v2.2.2     - indexer.metadata: Warn and skip
     incomplete entries from the journal
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 29 Jul 2022 11:52:13 +0000
 
 swh-indexer (2.2.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.2.1     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-07-29 10:56:57
     +0200)
   * Upstream changes:     - v2.2.1     - Normalize journal client
     indexer type names
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 29 Jul 2022 09:07:15 +0000
 
 swh-indexer (2.2.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.2.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-07-25 16:23:12
     +0200)
   * Upstream changes:     - v2.2.0     - cli: Add content mimetype
     indexer journal client support     - cli: Add fossology license
     indexer journal client support     - cli: Add extrinsic-metadata
     indexer journal client support     - docs: Fix incorrect terminology
     (term -> property)     - mapping: Fix inconsistent name     - Drop
     decommissioned content indexer: ctags, language
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 25 Jul 2022 14:33:50 +0000
 
 swh-indexer (2.1.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.1.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-07-21 10:23:44 +0200)
   * Upstream changes:     - v2.1.0     - * DirectoryIndexer: Remove
     incorrect assumption on object types     - * docs: Explain the
     indexation workflow for extrinsic metadata     - * docs: Update
     description of the metadata workflow     - * metadata_dictionary:
     Add mappings for pubspec.yaml     - * Add extrinsic metadata indexer
     - * Add GitHub metadata mapping     - * Refactor Mapping hierarchy
     - * cff: Add checks for value types
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 21 Jul 2022 08:32:23 +0000
 
 swh-indexer (2.0.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.0.2     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-06-22 12:32:41 +0200)
   * Upstream changes:     - v2.0.2     - * Fix mypy issue with swh-
     journal>=1.1.0     - * cff: Ignore invalid yaml files     - * npm:
     Add workaround for mangled package descriptions     - * npm: Fix
     crash when npm description is not a string
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 22 Jun 2022 10:40:25 +0000
 
 swh-indexer (2.0.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.0.1     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-06-10 10:35:15
     +0200)
   * Upstream changes:     - v2.0.1     - upgrades/134: Add missing index
     creation
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 10 Jun 2022 09:17:44 +0000
 
 swh-indexer (2.0.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 2.0.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2022-06-03 15:40:32
     +0200)
   * Upstream changes:     - v2.0.0     - Set current_version attribute
     to postgresql datastore     - Add support for indexing from head
     releases     - Replace RevisionMetadataIndexer with
     DirectoryMetadataIndexer     - Add support for running the server
     with 'postgresql' storage cls     - tests: Shorten definition of
     REVISION     - tests: Simplify definition of ORIGINS list     -
     tests: use stock pytest_postgresql factory function     - Rewrite
     origin_head.py as a normal function instead of an indexer     -
     Convert test_origin_head from unittest to pytest
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 03 Jun 2022 13:59:59 +0000
 
 swh-indexer (1.2.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 1.2.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-06-01 16:44:30 +0200)
   * Upstream changes:     - v1.2.0     - * cli: Add support for running
     "all" indexers in the journal client
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 01 Jun 2022 15:08:39 +0000
 
 swh-indexer (1.1.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 1.1.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-05-30 15:56:19 +0200)
   * Upstream changes:     - v1.1.0     - * Add support for indexing
     directly from the journal client     - * cff: Do not change
     yaml.SafeLoader globally     - * add missing sentry captures     - *
     Change misleading documentation in swh-indexer/cli.py     - * test
     and typing maintenance
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 30 May 2022 14:03:54 +0000
 
 swh-indexer (1.0.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 1.0.0     - (tagged by David Douard
     <david.douard@sdfa3.org> on 2022-02-24 17:35:56 +0100)
   * Upstream changes:     - v1.0.0
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 24 Feb 2022 16:42:39 +0000
 
 swh-indexer (0.8.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.8.2     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2022-01-12 13:53:22 +0100)
   * Upstream changes:     - v0.8.2     - * tests: Use
     TimestampWithTimezone.from_datetime() instead of the constructor
     - * docs: Use reference instead of absolute link
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 12 Jan 2022 12:56:56 +0000
 
 swh-indexer (0.8.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.8.1     - (tagged by Vincent SELLIER
     <vincent.sellier@softwareheritage.org> on 2021-12-21 16:23:37 +0100)
   * Upstream changes:     - v0.8.1     - Changelog:     - tag frozendict
     version to avoid segfaults on the ci
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 21 Dec 2021 15:28:27 +0000
 
 swh-indexer (0.8.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.8.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2021-05-28 16:57:47
     +0200)
   * Upstream changes:     - v0.8.0     - metadata_dictionary: Add
     mapping for CITATION.cff     - metadata/maven: Ignore ill-formed xml
     instead of failing     - metadata: Fix UnboundLocalError in edge
     case     - data/codemeta: sync with official codemeta repo     - Fix
     SingleFileMapping case sensitivity     - Use swh.core 0.14     -
     tox: Add sphinx environments to check sane doc build
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 28 May 2021 15:05:39 +0000
 
 swh-indexer (0.7.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.7.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2021-02-03 14:10:16
     +0100)
   * Upstream changes:     - v0.7.0     - Adapt
     origin_get_latest_visit_status according to latest api change
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 03 Feb 2021 13:15:37 +0000
 
 swh-indexer (0.6.4-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.6.4     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2021-02-01 15:06:04
     +0100)
   * Upstream changes:     - v0.6.4     - indexer: Remove pagination
     logic using stream_results() instead.     - ContentPartitionIndexer:
     Do not index the same content multiple times at once.     - Add a
     cli section in the doc     - test_journal_client_cli: Send
     production objects to journal     - test_journal_client: Migrate
     away from mocks     - tests: Use production backends within the
     indexer tests
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 01 Feb 2021 14:10:18 +0000
 
 swh-indexer (0.6.3-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.6.3     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-11-27 14:42:30
     +0100)
   * Upstream changes:     - v0.6.3     - storage.writer: Fix journal
     writer sanitizer function
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 27 Nov 2020 13:46:03 +0000
 
 swh-indexer (0.6.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.6.2     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-11-27 13:55:53
     +0100)
   * Upstream changes:     - v0.6.2     - BaseRow.unique_key: Don't crash
     when indexer_configuration_id is None.     -
     idx.storage.JournalWriter: pass value_sanitizer to
     get_journal_writer.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 27 Nov 2020 13:00:28 +0000
 
 swh-indexer (0.6.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.6.1     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-11-27 10:43:14
     +0100)
   * Upstream changes:     - v0.6.1     - Fix test within the debian
     package builds     - refactor tests to pytest
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 27 Nov 2020 09:49:35 +0000
 
 swh-indexer (0.6.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.6.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-11-26 17:08:03
     +0100)
   * Upstream changes:     - v0.6.0     - indexer.journal_client:
     Subscribe to OriginVisitStatus topic     -
     swh.indexer.cli.journal_client: ensure the minimal configuration
     exists     - Drop all deprecated uses of `args` in component
     factories     - Drop vcversioner from requirements     - Make the
     indexer storage write to the journal.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 26 Nov 2020 16:39:45 +0000
 
 swh-indexer (0.5.0-2~swh1) unstable-swh; urgency=medium
 
   * Move distutils package from python3-swh.indexer to python3-swh.indexer.storage.
 
  -- Nicolas Dandrimont <olasd@debian.org>  Wed, 18 Nov 2020 20:04:23 +0100
 
 swh-indexer (0.5.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.5.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2020-11-06 15:25:04 +0100)
   * Upstream changes:     - v0.5.0     - * Remove metadata deletion
     endpoints and algorithms     - * Remove
     conflict_update/policy_update option from BaseIndexer.run()     - *
     Remove conflict_update option from _add() endpoints.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 06 Nov 2020 14:28:05 +0000
 
 swh-indexer (0.4.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.4.2     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-10-30 17:22:22
     +0100)
   * Upstream changes:     - v0.4.2     - tests.conftest: Fix the indexer
     scheduler initialization     - indexer.cli: Fix missing retries_left
     parameter     - Rename sql files according to new conventions
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 30 Oct 2020 16:24:14 +0000
 
 swh-indexer (0.4.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.4.1     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-10-16 10:48:51
     +0200)
   * Upstream changes:     - v0.4.1     - test_cli: Remove unneeded
     config args parameter     - api.server: Align configuration
     structure with clients configuration     - storage.api.server: Add
     types to module and refactor tests
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 16 Oct 2020 08:59:09 +0000
 
 swh-indexer (0.4.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.4.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-10-15 18:17:59
     +0200)
   * Upstream changes:     - v0.4.0     - swh.indexer.storage: Unify
     get_indexer_storage function with others
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 15 Oct 2020 16:19:01 +0000
 
 swh-indexer (0.3.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.3.0     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2020-10-08 13:33:02 +0200)
   * Upstream changes:     - v0.3.0     - * Make indexer-storage
     endpoints use attr-based classes instead of dicts     - * Add more
     typing to indexers and their tests
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 08 Oct 2020 11:35:50 +0000
 
 swh-indexer (0.2.4-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.2.4     - (tagged by David Douard
     <david.douard@sdfa3.org> on 2020-09-25 12:49:04 +0200)
   * Upstream changes:     - v0.2.4
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 25 Sep 2020 10:51:28 +0000
 
 swh-indexer (0.2.3-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.2.3     - (tagged by David Douard
     <david.douard@sdfa3.org> on 2020-09-11 15:12:01 +0200)
   * Upstream changes:     - v0.2.3
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 11 Sep 2020 13:15:41 +0000
 
 swh-indexer (0.2.2-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.2.2     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-09-04 13:21:19
     +0200)
   * Upstream changes:     - v0.2.2     - metadata: Adapt to latest
     storage revision_get change     - Tell pytest not to recurse in
     dotdirs.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 04 Sep 2020 11:33:41 +0000
 
 swh-indexer (0.2.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.2.1     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2020-08-20 12:59:53 +0200)
   * Upstream changes:     - v0.2.1     - * indexer.rehash: Adapt
     content_get_metadata call to content_get     - * origin_head: Use
     snapshot_get_all_branches instead of snapshot_get.     - * Import
     SortedList, db_transaction_generator, and db_transaction from swh-
     core instead of swh-storage.     - * tests: remove invalid assertion
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 20 Aug 2020 11:03:58 +0000
 
 swh-indexer (0.2.0-1~swh2) unstable-swh; urgency=medium
 
   * Bump dependencies
 
  -- Antoine R. Dumont (@ardumont) <ardumont@softwareheritage.org>  Wed, 06 Aug 2020 13:28:00 +0200
 
 swh-indexer (0.2.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.2.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-08-06 15:12:44
     +0200)
   * Upstream changes:     - v0.2.0     - Make content indexer work on
     partition of ids
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 06 Aug 2020 13:14:35 +0000
 
 swh-indexer (0.1.1-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.1.1     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-07-28 12:42:19
     +0200)
   * Upstream changes:     - v0.1.1     - setup.py: Migrate from
     vcversioner to setuptools-scm     - MANIFEST: Include missing
     conftest.py requirement     - metadata: Update
     swh.storage.origin_get call to latest api change     - Drop
     unsupported "validate" proxy     - tests: Drop deprecated
     storage.origin_add_one use     - Drop useless use of pifpaf     -
     Clean up the swh.scheduler and swh.storage pytest plugin imports
     - tests: Drop obsolete origin visit fields
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 28 Jul 2020 10:44:54 +0000
 
 swh-indexer (0.1.0-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.1.0     - (tagged by Antoine R. Dumont
     (@ardumont) <ardumont@softwareheritage.org> on 2020-06-23 15:44:15
     +0200)
   * Upstream changes:     - v0.1.0     - origin_head: Retrieve snapshot
     out of the last visit status     - Fix tests according to latest
     internal api changes
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 23 Jun 2020 13:46:23 +0000
 
 swh-indexer (0.0.171-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.171     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-04-23 16:46:52
     +0200)
   * Upstream changes:     - v0.0.171     - cli: Adapt journal client
     instantiation according to latest change     - codemeta: Add
     compatibility with PyLD >= 2.0.0.     - setup: Update the minimum
     required runtime python3 version     - Add a pyproject.toml file to
     target py37 for black     - Enable black     - test: make test data
     properly typed     - indexer.cli.journal_client: Simplify the
     journal client call     - Remove type from origin_add calls     -
     Rename --max-messages to --stop-after-objects.     - tests: Migrate
     to latest swh-storage api change
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 23 Apr 2020 14:49:17 +0000
 
 swh-indexer (0.0.170-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.170     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-08 09:57:39
     +0100)
   * Upstream changes:     - v0.0.170     - indexer.metadata: Make
     compatible old task format
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Sun, 08 Mar 2020 09:03:59 +0000
 
 swh-indexer (0.0.169-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.169     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-06 15:19:21
     +0100)
   * Upstream changes:     - v0.0.169     - storage: Add @timed metrics
     on remaining indexer storage endpoints     - indexer.storage: Use
     the correct metrics module     - idx.storage: Add time and counter
     metric to idx_configuration_add     - indexer.storage: Remove
     redundant calls to send_metric     - indexer: Fix mypy issues
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 06 Mar 2020 14:24:50 +0000
 
 swh-indexer (0.0.168-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.168     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-05 15:48:32
     +0100)
   * Upstream changes:     - v0.0.168     - mimetype: Make the parsing
     more resilient     - storage.fossology_license_add: Fix one insert
     query too many     - tests: Migrate some tests to pytest
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 05 Mar 2020 14:52:27 +0000
 
 swh-indexer (0.0.167-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.167     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-04 16:33:20
     +0100)
   * Upstream changes:     - v0.0.167     - indexer (revision, origin):
     Fix indexer summary to output a status
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 04 Mar 2020 15:37:59 +0000
 
 swh-indexer (0.0.166-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.166     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2020-03-04 15:46:37 +0100)
   * Upstream changes:     - v0.0.166     - * Fix merging documents with
     @list elements.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 04 Mar 2020 14:50:54 +0000
 
 swh-indexer (0.0.165-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.165     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-04 15:29:52
     +0100)
   * Upstream changes:     - v0.0.165     - indexers: Fix summary
     computation for range indexers     - tests: Use assertEqual instead
     of deprecated assertEquals
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 04 Mar 2020 14:33:09 +0000
 
 swh-indexer (0.0.164-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.164     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-04 13:52:15
     +0100)
   * Upstream changes:     - v0.0.164     - range-indexers: Fix hard-
     coded summary key value     - indexers: Improve
     persist_index_computations type     - indexer.metadata: Fix wrong
     update
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 04 Mar 2020 13:00:18 +0000
 
 swh-indexer (0.0.163-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.163     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-03-04 11:26:56
     +0100)
   * Upstream changes:     - v0.0.163     - Make indexers return a
     summary of their actions     - swh.indexer.storage: Add metrics to
     add/del endpoints     - indexer.storage: Make add/del endpoints sum
     up added objects count     - indexer: Remove unused next_step
     pattern
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 04 Mar 2020 10:31:03 +0000
 
 swh-indexer (0.0.162-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.162     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-02-27 11:01:29
     +0100)
   * Upstream changes:     - v0.0.162     - fossology_license: Improve
     add query endpoint     - pgstorage: Empty temp tables instead of
     dropping them     - indexer.metadata: Fix edge case on unknown
     origin
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 27 Feb 2020 10:09:36 +0000
 
 swh-indexer (0.0.161-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.161     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-02-25 12:07:39
     +0100)
   * Upstream changes:     - v0.0.161     - sql/128: Add content_mimetype
     index     - storage.db: Improve content range queries to actually
     finish     - Add a new IndexerStorageArgumentException class, for
     exceptions caused by the client.     - Use swh-storage validation
     proxy.     - Fix type errors with hypothesis 5.5     - Add type
     annotations to indexer classes
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 25 Feb 2020 11:20:51 +0000
 
 swh-indexer (0.0.160-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.160     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-02-05 18:13:16
     +0100)
   * Upstream changes:     - v0.0.160     - Fix missing import
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 05 Feb 2020 17:28:18 +0000
 
 swh-indexer (0.0.159-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.159     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2020-02-05 16:01:03
     +0100)
   * Upstream changes:     - v0.0.159     - Monkey-patch backend classes
     instead of 'get_storage' functions.     - Fix DeprecationWarning
     about get_storage args.     - Move IndexerStorage documentation and
     endpoint paths to a new IndexerStorageInterface class.     -
     conftest: Use module's `get_<storage-backend>` to instantiate
     backend     - docs: Fix sphinx warnings     - Fix merge_documents to
     work with input document with an @id.     - Fix support of VCSs
     whose HEAD branch is an alias.     - Fix type of 'author' in gemspec
     mapping output.     - Fix test_origin_metadata mistakenly broken by
     e50660efca     - Fix several typos reported by pre-commit hooks     -
     Add a pre-commit config file     - Remove unused property-based test
     environment     - Migrate tox.ini to extras = xxx instead of deps =
     .[testing]     - Merge tox test environments     - Drop version
     constraint on pytest     - Include all requirements in MANIFEST.in
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 05 Feb 2020 15:09:42 +0000
 
 swh-indexer (0.0.158-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.158     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-11-20 10:26:59
     +0100)
   * Upstream changes:     - v0.0.158     - Re-enable tests for the in-
     memory storage.     - Truncate result list instead of doing a copy.
     - journal client: add support for new origin_visit schema.     - Fix
     alter table rename column syntax on 126->127 upgrade script
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 20 Nov 2019 09:30:37 +0000
 
 swh-indexer (0.0.157-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.157     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-11-08 16:33:36 +0100)
   * Upstream changes:     - v0.0.157     - * migrate storage tests to
     pytest     - * proper pagination for
     IndexerStorage.origin_intrinsic_metadata_search_by_producer
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 08 Nov 2019 15:36:48 +0000
 
 swh-indexer (0.0.156-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.156     - (tagged by Stefano Zacchiroli
     <zack@upsilon.cc> on 2019-11-05 17:36:11 +0100)
   * Upstream changes:     - v0.0.156     - * update indexer for storage
     0.0.156     - * cli: fix max-message handling in the journal-client
     command     - * tests: fix test_metadata.py for frozen entities in
     swh.model.model     - * tests: update tests for storage>=0.0.155
     - * test_metadata typing: use type-specific mappings instead of cast
     - * storage/db.py: drop unused format arg regconfig from query     -
     * typing: minimal changes to make a no-op mypy run pass
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 05 Nov 2019 16:45:10 +0000
 
 swh-indexer (0.0.155-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.155     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-10-15 14:51:28 +0200)
   * Upstream changes:     - v0.0.155     - * Avoid spamming logs with
     processed %d messages every message     - * tox.ini: Fix py3
     environment to use packaged tests     - * Remove indirection
     swh.indexer.storage.api.wsgi to start server     - * Add a command-
     line tool to run metadata translation.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 15 Oct 2019 12:55:33 +0000
 
 swh-indexer (0.0.154-1~swh2) unstable-swh; urgency=medium
 
   * Force pg_ctl path
 
  -- Nicolas Dandrimont <olasd@debian.org>  Mon, 07 Oct 2019 16:42:08 +0200
 
 swh-indexer (0.0.154-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.154     - (tagged by Nicolas Dandrimont
     <nicolas@dandrimont.eu> on 2019-10-07 16:34:20 +0200)
   * Upstream changes:     - Release swh.indexer v0.0.154     - Remove
     old scheduler compat code     - Clean up CLI aliases     - Port to
     python-magic instead of file_magic
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 07 Oct 2019 14:38:47 +0000
 
 swh-indexer (0.0.153-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.153     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-09-11 11:46:41
     +0200)
   * Upstream changes:     - v0.0.153     - indexer-storage: Send smaller
     batches to origin_get     - Update
     origin_url/from_revision/metadata_tsvector when conflict_update=True
     - Remove concept of 'minimal set' of metadata     - npm: Fix crash
     on invalid 'author' field     - api/client: use RPCClient instead of
     deprecated SWHRemoteAPI     - api/server: use RPCServerApp instead
     of deprecated SWHServerAPIApp     - tests/utils: Fix various test
     data model issues failing validation
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 11 Sep 2019 09:50:58 +0000
 
 swh-indexer (0.0.152-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.152     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-07-19 11:15:41 +0200)
   * Upstream changes:     - Send smaller batches to revision_get
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 19 Jul 2019 09:20:34 +0000
 
 swh-indexer (0.0.151-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.151     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-07-03 17:58:32 +0200)
   * Upstream changes:     - v0.0.151     - Fix key names in the journal
     client; it crashed in prod.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 03 Jul 2019 16:03:07 +0000
 
 swh-indexer (0.0.150-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.150     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-07-03 12:09:43
     +0200)
   * Upstream changes:     - v0.0.150     - indexer.cli: Drop unused
     extra alias `--consumer-id` flag
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 03 Jul 2019 10:20:46 +0000
 
 swh-indexer (0.0.149-1~swh2) unstable-swh; urgency=medium
 
   * No-change: Bump dependency version
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 03 Jul 2019 10:44:12 +0200
 
 swh-indexer (0.0.149-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.149     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-07-02 18:11:12
     +0200)
   * Upstream changes:     - v0.0.149     - swh.indexer.cli: Fix
     get_journal_client api call     - sql/upgrades/125: Fix migration
     script
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 02 Jul 2019 16:26:50 +0000
 
 swh-indexer (0.0.148-1~swh3) unstable-swh; urgency=medium
 
   * Upstream release 0.0.148: Update version dependency
 
  -- Antoine Romain Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 01 Jul 2019 01:50:29 +0100
 
 swh-indexer (0.0.148-1~swh2) unstable-swh; urgency=medium
 
   * Upstream release 0.0.148
 
  -- Antoine Romain Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 01 Jul 2019 01:50:29 +0100
 
 swh-indexer (0.0.148-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.148     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-07-01 12:21:32
     +0200)
   * Upstream changes:     - v0.0.148     - Manipulate origin URLs
     instead of origin ids     - journal: create tasks for multiple
     origins     - Tests: Improvements
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 01 Jul 2019 10:34:26 +0000
 
 swh-indexer (0.0.147-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.147     - (tagged by Antoine Lambert
     <antoine.lambert@inria.fr> on 2019-05-23 11:03:02 +0200)
   * Upstream changes:     - version 0.0.147
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 23 May 2019 09:11:05 +0000
 
 swh-indexer (0.0.146-1~swh2) unstable-swh; urgency=medium
 
   * Remove hypothesis directory
 
  -- Nicolas Dandrimont <olasd@debian.org>  Thu, 18 Apr 2019 18:29:09 +0200
 
 swh-indexer (0.0.146-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.146     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-04-11 11:08:29 +0200)
   * Upstream changes:     - Better explain what the 'string fields' are.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 11 Apr 2019 09:47:24 +0000
 
 swh-indexer (0.0.145-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.145     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-03-15 11:18:25 +0100)
   * Upstream changes:     - Add support for keywords in PKG-INFO.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 15 Mar 2019 11:34:53 +0000
 
 swh-indexer (0.0.144-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.144     - (tagged by Thibault Allançon
     <tallancon@gmail.com> on 2019-03-07 08:16:49 +0100)
   * Upstream changes:     - Fix heterogeneity of names in metadata
     tables
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 14 Mar 2019 13:30:44 +0000
 
 swh-indexer (0.0.143-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.143     - (tagged by Thibault Allançon
     <tallancon@gmail.com> on 2019-03-12 10:18:37 +0100)
   * Upstream changes:     - Use hashutil.MultiHash in
     swh.indexer.tests.test_utils.fill_storage     - Summary: Closes
     T1448     - Reviewers: #reviewers     - Subscribers: swh-public-ci
     - Maniphest Tasks: T1448     - Differential Revision:
     https://forge.softwareheritage.org/D1235
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 13 Mar 2019 10:24:37 +0000
 
 swh-indexer (0.0.142-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.142     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-03-01 14:19:05 +0100)
   * Upstream changes:     - Skip useless requests.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 01 Mar 2019 13:26:06 +0000
 
 swh-indexer (0.0.141-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.141     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-03-01 10:59:54 +0100)
   * Upstream changes:     - Prevent origin metadata indexer from writing
     empty records
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 01 Mar 2019 10:10:56 +0000
 
 swh-indexer (0.0.140-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.140     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-02-25 10:38:52 +0100)
   * Upstream changes:     - Drop the 'context' and 'type' config of
     metadata indexers.     - They are both ignored already.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 25 Feb 2019 10:40:10 +0000
 
 swh-indexer (0.0.139-1~swh2) unstable-swh; urgency=low
 
   * New release fixing debian build
 
  -- Antoine Romain Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 22 Feb 2019 16:27:47 +0100
 
 swh-indexer (0.0.139-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.139     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-02-22 15:53:22
     +0100)
   * Upstream changes:     - v0.0.139     - Clean up no longer used tasks
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 22 Feb 2019 14:59:40 +0000
 
 swh-indexer (0.0.138-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.138     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-02-22 15:30:30 +0100)
   * Upstream changes:     - Make the 'config' argument of
     OriginMetadaIndexer optional again.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 22 Feb 2019 14:37:35 +0000
 
 swh-indexer (0.0.137-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.137     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-02-22 10:59:53
     +0100)
   * Upstream changes:     - v0.0.137     - swh.indexer.storage.api.wsgi:
     Open production wsgi entrypoint     - swh.indexer.cli: Move dev app
     entrypoint in dedicated cli     - indexer.storage: Make server load
     explicit configuration and check     - config: use already loaded
     swh config, if any, when instantiating an Indexer     - api: Add
     support for filtering by tool_id to
     origin_intrinsic_metadata_search_by_producer.     - api: Add storage
     endpoint to search metadata by mapping.     - runtime: Remove
     implicit configuration from the metadata indexers.     - debian:
     Remove debian packaging from master branch     - docs: Update
     missing documentation
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 22 Feb 2019 10:11:29 +0000
 
 swh-indexer (0.0.136-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.136     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-02-14 17:09:00 +0100)
   * Upstream changes:     - Don't send 'None' as a revision id to
     storage.revision_get.     - This error wasn't caught before because
     the in-mem storage     - accepts None values, but the pg storage
     doesn't.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 14 Feb 2019 16:22:41 +0000
 
 swh-indexer (0.0.135-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.135     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-02-14 14:45:24 +0100)
   * Upstream changes:     - Fix deduplication of origins when persisting
     origin intrinsic metadata.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 14 Feb 2019 14:32:55 +0000
 
 swh-indexer (0.0.134-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.134     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-02-13 23:46:44
     +0100)
   * Upstream changes:     - v0.0.134     - package: Break dependency of
     swh.indexer.storage on swh.indexer.     - api/server: Do not read
     configuration at each request     - metadata: Fix gemspec test     -
     metadata: Prevent OriginMetadataIndexer from sending duplicate     -
     revisions to revision_metadata_add.     - test: Fix bugs found by
     hypothesis.     - test: Use hypothesis to generate adversarial
     inputs.     - Add more type checks in metadata dictionary.     - Add
     checks in the idx_storage that the same content/rev/orig is not     -
     present twice in the new data.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 14 Feb 2019 09:16:15 +0000
 
 swh-indexer (0.0.133-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.133     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-02-12 10:28:01
     +0100)
   * Upstream changes:     - v0.0.133     - Migrate BaseDB api calls from
     core to storage     - Improve storage api calls using latest storage
     api     - OriginIndexer: Refactoring     - tests: Refactoring     -
     metadata search: Use index     - indexer metadata: Provide stats per
     origin     - indexer metadata: Update mapping column     - indexer
     metadata: Improve and fix issues
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 12 Feb 2019 09:34:43 +0000
 
 swh-indexer (0.0.132-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.132     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-01-30 15:03:14
     +0100)
   * Upstream changes:     - v0.0.132     - swh/indexer/tasks: Fix range
     indexer tasks     - Maven: Add support for empty XML nodes.     -
     Add support for alternative call format for Gem::Specification.new.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 30 Jan 2019 14:09:48 +0000
 
 swh-indexer (0.0.131-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.131     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-01-30 10:56:43
     +0100)
   * Upstream changes:     - v0.0.131     - fix pep8 violations     - fix
     misspellings
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Wed, 30 Jan 2019 10:01:47 +0000
 
 swh-indexer (0.0.129-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.129     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-01-29 14:11:22 +0100)
   * Upstream changes:     - Fix missing config file name change.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 29 Jan 2019 13:34:17 +0000
 
 swh-indexer (0.0.128-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.128     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-01-25 15:22:52 +0100)
   * Upstream changes:     - Make metadata indexers store the mappings
     used to translate metadata.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 29 Jan 2019 12:18:16 +0000
 
 swh-indexer (0.0.127-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.127     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-01-15 15:56:49 +0100)
   * Upstream changes:     - Prevent repository normalization from
     crashing on malformed input.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Tue, 15 Jan 2019 16:20:32 +0000
 
 swh-indexer (0.0.126-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.126     - (tagged by Valentin Lorentz
     <vlorentz@softwareheritage.org> on 2019-01-14 11:42:52 +0100)
   * Upstream changes:     - Don't call OriginHeadIndexer.next_step when
     there is no revision.
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Mon, 14 Jan 2019 10:57:34 +0000
 
 swh-indexer (0.0.125-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.125     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-01-11 12:01:42
     +0100)
   * Upstream changes:     - v0.0.125     - Add journal client that
     listens for origin visits and schedules     - OriginHead     - Fix
     tests to work with the new version of swh.storage
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Fri, 11 Jan 2019 11:08:51 +0000
 
 swh-indexer (0.0.124-1~swh1) unstable-swh; urgency=medium
 
   * New upstream release 0.0.124     - (tagged by Antoine R. Dumont
     (@ardumont) <antoine.romain.dumont@gmail.com> on 2019-01-08 14:09:32
     +0100)
   * Upstream changes:     - v0.0.124     - indexer: Fix type check on
     indexing result
 
  -- Software Heritage autobuilder (on jenkins-debian1) <jenkins@jenkins-debian1.internal.softwareheritage.org>  Thu, 10 Jan 2019 17:12:07 +0000
 
 swh-indexer (0.0.118-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.118
   * metadata-indexer: Fix setup initialization
   * tests: Refactoring
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 30 Nov 2018 14:50:52 +0100
 
 swh-indexer (0.0.67-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.67
   * mimetype: Migrate to indexed data as text
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 28 Nov 2018 11:35:37 +0100
 
 swh-indexer (0.0.66-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.66
   * range-indexer: Stream indexing range computations
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 27 Nov 2018 11:48:24 +0100
 
 swh-indexer (0.0.65-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.65
   * Fix revision metadata indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 26 Nov 2018 19:30:48 +0100
 
 swh-indexer (0.0.64-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.64
   * indexer: Fix mixed identifier encodings issues
   * Add missing config filename for origin intrinsic metadata indexer.
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 26 Nov 2018 12:20:01 +0100
 
 swh-indexer (0.0.63-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.63
   * Make the OriginMetadataIndexer fetch rev metadata from the storage
   * instead of getting them via the scheduler.
   * Make the 'result_name' key of 'next_step' optional.
   * Add missing return.
   * doc: update index to match new swh-doc format
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 23 Nov 2018 17:56:10 +0100
 
 swh-indexer (0.0.62-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.62
   * metadata indexer: Add empty tool configuration
   * Add fulltext search on origin intrinsic metadata
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 23 Nov 2018 14:25:55 +0100
 
 swh-indexer (0.0.61-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.61
   * indexer: Fix origin indexer's default arguments
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 21 Nov 2018 16:01:50 +0100
 
 swh-indexer (0.0.60-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.60
   * origin_head: Make next step optional
   * tests: Increase coverage
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 21 Nov 2018 12:33:13 +0100
 
 swh-indexer (0.0.59-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.59
   * fossology license: Fix issue on license computation
   * Improve docstrings
   * Fix pep8 violations
   * Increase coverage on content indexers
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 20 Nov 2018 14:27:20 +0100
 
 swh-indexer (0.0.58-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.58
   * Add missing default configuration for fossology license indexer
   * tests: Remove dead code
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 20 Nov 2018 12:06:56 +0100
 
 swh-indexer (0.0.57-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.57
   * storage: Open new endpoint on fossology license range retrieval
   * indexer: Open new fossology license range indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 20 Nov 2018 11:44:57 +0100
 
 swh-indexer (0.0.56-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.56
   * storage.api: Open new endpoints (mimetype range, fossology range)
   * content indexers: Open mimetype and fossology range indexers
   * Remove orchestrator modules
   * tests: Improve coverage
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 19 Nov 2018 11:56:06 +0100
 
 swh-indexer (0.0.55-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.55
   * swh.indexer: Let task reschedule itself through the scheduler
   * Use swh.scheduler instead of celery leaking all around
   * swh.indexer.orchestrator: Fix orchestrator initialization step
   * swh.indexer.tasks: Fix type error when no result or list result
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 29 Oct 2018 10:41:54 +0100
 
 swh-indexer (0.0.54-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.54
   * swh.indexer.tasks: Fix task to use the scheduler's
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 25 Oct 2018 20:13:51 +0200
 
 swh-indexer (0.0.53-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.53
   * swh.indexer.rehash: Migrate to latest swh.model.hashutil.MultiHash
   * indexer: Add the origin intrinsic metadata indexer
   * indexer: Add OriginIndexer and OriginHeadIndexer.
   * indexer.storage: Add the origin intrinsic metadata storage database
   * indexer.storage: Autogenerate the Indexer Storage HTTP API.
   * setup: prepare for pypi upload
   * tests: Add a tox file
   * tests: migrate to pytest
   * tests: Add tests around celery stack
   * docs: Improve documentation and reuse README in generated
     documentation
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 25 Oct 2018 19:03:56 +0200
 
 swh-indexer (0.0.52-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.52
   * swh.indexer.storage: Refactor fossology license get (first external
   * contribution, cf. /CONTRIBUTORS)
   * swh.indexer.storage: Fix typo in invariable name metadata
   * swh.indexer.storage: No longer use temp table when reading data
   * swh.indexer.storage: Clean up unused import
   * swh.indexer.storage: Remove dead entry points origin_metadata*
   * swh.indexer.storage: Update docstrings information and format
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 13 Jun 2018 11:20:40 +0200
 
 swh-indexer (0.0.51-1~swh1) unstable-swh; urgency=medium
 
   * Release swh.indexer v0.0.51
   * Update for new db_transaction{,_generator}
 
  -- Nicolas Dandrimont <nicolas@dandrimont.eu>  Tue, 05 Jun 2018 14:10:39 +0200
 
 swh-indexer (0.0.50-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.50
   * swh.indexer.api.client: Permit to specify the query timeout option
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 24 May 2018 12:19:06 +0200
 
 swh-indexer (0.0.49-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.49
   * test_storage: Instantiate the tools during tests' setUp phase
   * test_storage: Deallocate storage during teardown step
   * test_storage: Make storage test fixture connect to postgres itself
   * storage.api.server: Only instantiate storage backend once per import
   * Use thread-aware psycopg2 connection pooling for database access
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 14 May 2018 11:09:30 +0200
 
 swh-indexer (0.0.48-1~swh1) unstable-swh; urgency=medium
 
   * Release swh.indexer v0.0.48
   * Update for new swh.storage
 
  -- Nicolas Dandrimont <nicolas@dandrimont.eu>  Sat, 12 May 2018 18:30:10 +0200
 
 swh-indexer (0.0.47-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.47
   * d/control: Fix runtime typo in packaging dependency
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 07 Dec 2017 16:54:49 +0100
 
 swh-indexer (0.0.46-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.46
   * Split swh-indexer packages in 2 python3-swh.indexer.storage and
   * python3-swh.indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 07 Dec 2017 16:18:04 +0100
 
 swh-indexer (0.0.45-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.45
   * Fix usual error raised when deploying
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 07 Dec 2017 15:01:01 +0100
 
 swh-indexer (0.0.44-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.44
   * swh.indexer: Make indexer use their own storage
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 07 Dec 2017 13:20:44 +0100
 
 swh-indexer (0.0.43-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.43
   * swh.indexer.mimetype: Work around problem in detection
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 29 Nov 2017 10:26:11 +0100
 
 swh-indexer (0.0.42-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.42
   * swh.indexer: Make indexers register tools in prepare method
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 24 Nov 2017 11:26:03 +0100
 
 swh-indexer (0.0.41-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.41
   * mimetype: Use magic library api instead of parsing `file` cli output
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 20 Nov 2017 13:05:29 +0100
 
 swh-indexer (0.0.39-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.39
   * swh.indexer.producer: Fix argument to match the abstract definition
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 19 Oct 2017 10:03:44 +0200
 
 swh-indexer (0.0.38-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.38
   * swh.indexer.indexer: Fix argument to match the abstract definition
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 18 Oct 2017 19:57:47 +0200
 
 swh-indexer (0.0.37-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.37
   * swh.indexer.indexer: Fix argument to match the abstract definition
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 18 Oct 2017 18:59:42 +0200
 
 swh-indexer (0.0.36-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.36
   * packaging: Cleanup
   * codemeta: Adding codemeta.json file to document metadata
   * swh.indexer.mimetype: Fix edge case regarding empty raw content
   * docs: sanitize docstrings for sphinx documentation generation
   * swh.indexer.metadata: Add RevisionMetadataIndexer
   * swh.indexer.metadata: Add ContentMetadataIndexer
   * swh.indexer: Refactor base class to improve inheritance
   * swh.indexer.metadata: First draft of the metadata content indexer
   * for npm (package.json)
   * swh.indexer.tests: Added tests for language indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 18 Oct 2017 16:24:24 +0200
 
 swh-indexer (0.0.35-1~swh1) unstable-swh; urgency=medium
 
   * Release swh.indexer 0.0.35
   * Update tasks to new swh.scheduler API
 
  -- Nicolas Dandrimont <nicolas@dandrimont.eu>  Mon, 12 Jun 2017 18:02:04 +0200
 
 swh-indexer (0.0.34-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.34
   * Fix unbound local error on edge case
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 07 Jun 2017 11:23:29 +0200
 
 swh-indexer (0.0.33-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.33
   * language indexer: Improve edge case policy
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 07 Jun 2017 11:02:47 +0200
 
 swh-indexer (0.0.32-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.32
   * Update fossology license to use the latest swh-storage
   * Improve language indexer to deal with potential error on bad
   * chunking
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 06 Jun 2017 18:13:40 +0200
 
 swh-indexer (0.0.31-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.31
   * Reduce log verbosity on language indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 02 Jun 2017 19:08:52 +0200
 
 swh-indexer (0.0.30-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.30
   * Fix wrong default configuration
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 02 Jun 2017 18:01:27 +0200
 
 swh-indexer (0.0.29-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.29
   * Update indexer to resolve indexer configuration identifier
   * Adapt language indexer to use partial raw content
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 02 Jun 2017 16:21:27 +0200
 
 swh-indexer (0.0.28-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.28
   * Add error resilience to fossology indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 22 May 2017 12:57:55 +0200
 
 swh-indexer (0.0.27-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.27
   * swh.indexer.language: Incremental encoding detection
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 17 May 2017 18:04:27 +0200
 
 swh-indexer (0.0.26-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.26
   * swh.indexer.orchestrator: Add batch size option per indexer
   * Log caught exception in a unified manner
   * Add rescheduling option (not by default) on rehash + indexers
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 17 May 2017 14:08:07 +0200
 
 swh-indexer (0.0.25-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.25
   * Add reschedule on error parameter for indexers
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 12 May 2017 12:13:15 +0200
 
 swh-indexer (0.0.24-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.24
   * Make rehash indexer more resilient to errors by rescheduling
     contents
   * in error (be it reading or updating problems)
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 04 May 2017 14:22:43 +0200
 
 swh-indexer (0.0.23-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.23
   * Improve producer to optionally make it synchronous
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 03 May 2017 15:29:44 +0200
 
 swh-indexer (0.0.22-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.22
   * Improve mimetype indexer implementation
   * Make the chaining option in the mimetype indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 02 May 2017 16:31:14 +0200
 
 swh-indexer (0.0.21-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.21
   * swh.indexer.rehash: Actually make the worker log
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 02 May 2017 14:28:55 +0200
 
 swh-indexer (0.0.20-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.20
   * swh.indexer.rehash:
   * Improve reading from objstorage only when needed
   * Fix empty file use case (which was skipped)
   * Add logging
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 28 Apr 2017 09:39:09 +0200
 
 swh-indexer (0.0.19-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.19
   * Fix rehash indexer's default configuration file
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 27 Apr 2017 19:17:20 +0200
 
 swh-indexer (0.0.18-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.18
   * Add new rehash indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 26 Apr 2017 15:23:02 +0200
 
 swh-indexer (0.0.17-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.17
   * Add information on indexer tools (T610)
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 02 Dec 2016 18:32:54 +0100
 
 swh-indexer (0.0.16-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.16
   * bug fixes
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 15 Nov 2016 19:31:52 +0100
 
 swh-indexer (0.0.15-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.15
   * Improve message producer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 15 Nov 2016 18:16:42 +0100
 
 swh-indexer (0.0.14-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.14
   * Update package dependency on fossology-nomossa
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Tue, 15 Nov 2016 14:13:41 +0100
 
 swh-indexer (0.0.13-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.13
   * Add new license indexer
   * ctags indexer: align behavior with other indexers regarding the
   * conflict update policy
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Mon, 14 Nov 2016 14:13:34 +0100
 
 swh-indexer (0.0.12-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.12
   * Add runtime dependency on universal-ctags
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 04 Nov 2016 13:59:59 +0100
 
 swh-indexer (0.0.11-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.11
   * Remove dependency on exuberant-ctags
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 03 Nov 2016 16:13:26 +0100
 
 swh-indexer (0.0.10-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.10
   * Add ctags indexer
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 20 Oct 2016 16:12:42 +0200
 
 swh-indexer (0.0.9-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.9
   * d/control: Bump dependency to latest python3-swh.storage api
   * mimetype: Use the charset to filter out data
   * orchestrator: Separate 2 distincts orchestrators (one for all
   * contents, one for text contents)
   * mimetype: once index computed, send text contents to text
     orchestrator
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 13 Oct 2016 15:28:17 +0200
 
 swh-indexer (0.0.8-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.8
   * Separate configuration file per indexer (no need for language)
   * Rename module file_properties to mimetype consistently with other
   * layers
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Sat, 08 Oct 2016 11:46:29 +0200
 
 swh-indexer (0.0.7-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.7
   * Adapt indexer language and mimetype to store result in storage.
   * Clean up obsolete code
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Sat, 08 Oct 2016 10:26:08 +0200
 
 swh-indexer (0.0.6-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.6
   * Fix multiple issues on production
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 30 Sep 2016 17:00:11 +0200
 
 swh-indexer (0.0.5-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.5
   * Fix debian/control dependency issue
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 30 Sep 2016 16:06:20 +0200
 
 swh-indexer (0.0.4-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.4
   * Upgrade dependencies issues
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 30 Sep 2016 16:01:52 +0200
 
 swh-indexer (0.0.3-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.3
   * Add encoding detection
   * Use encoding to improve language detection
   * bypass language detection for binary files
   * bypass ctags for binary files or decoding failure file
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Fri, 30 Sep 2016 12:30:11 +0200
 
 swh-indexer (0.0.2-1~swh1) unstable-swh; urgency=medium
 
   * v0.0.2
   * Provide one possible sha1's name for the multiple tools to ease
   * information extrapolation
   * Fix debian package dependency issue
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Thu, 29 Sep 2016 21:45:44 +0200
 
 swh-indexer (0.0.1-1~swh1) unstable-swh; urgency=medium
 
   * Initial release
   * v0.0.1
   * First implementation on poc
 
  -- Antoine R. Dumont (@ardumont) <antoine.romain.dumont@gmail.com>  Wed, 28 Sep 2016 23:40:13 +0200
diff --git a/swh.indexer.egg-info/PKG-INFO b/swh.indexer.egg-info/PKG-INFO
index 7c5cbd3..4912824 100644
--- a/swh.indexer.egg-info/PKG-INFO
+++ b/swh.indexer.egg-info/PKG-INFO
@@ -1,71 +1,71 @@
 Metadata-Version: 2.1
 Name: swh.indexer
-Version: 2.8.0
+Version: 2.9.0
 Summary: Software Heritage Content Indexer
 Home-page: https://forge.softwareheritage.org/diffusion/78/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-indexer
 Project-URL: Documentation, https://docs.softwareheritage.org/devel/swh-indexer/
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Requires-Python: >=3.7
 Description-Content-Type: text/markdown
 Provides-Extra: testing
 License-File: LICENSE
 License-File: AUTHORS
 
 swh-indexer
 ============
 
 Tools to compute multiple indexes on SWH's raw contents:
 - content:
   - mimetype
   - ctags
   - language
   - fossology-license
   - metadata
 - revision:
   - metadata
 
 An indexer is in charge of:
 - looking up objects
 - extracting information from those objects
 - store those information in the swh-indexer db
 
 There are multiple indexers working on different object types:
   - content indexer: works with content sha1 hashes
   - revision indexer: works with revision sha1 hashes
   - origin indexer: works with origin identifiers
 
 Indexation procedure:
 - receive batch of ids
 - retrieve the associated data depending on object type
 - compute for that object some index
 - store the result to swh's storage
 
 Current content indexers:
 
 - mimetype (queue swh_indexer_content_mimetype): detect the encoding
   and mimetype
 
 - language (queue swh_indexer_content_language): detect the
   programming language
 
 - ctags (queue swh_indexer_content_ctags): compute tags information
 
 - fossology-license (queue swh_indexer_fossology_license): compute the
   license
 
 - metadata: translate file into translated_metadata dict
 
 Current revision indexers:
 
 - metadata: detects files containing metadata and retrieves translated_metadata
   in content_metadata table in storage or run content indexer to translate
   files.
diff --git a/swh.indexer.egg-info/SOURCES.txt b/swh.indexer.egg-info/SOURCES.txt
index 1cfefa4..c1d345d 100644
--- a/swh.indexer.egg-info/SOURCES.txt
+++ b/swh.indexer.egg-info/SOURCES.txt
@@ -1,169 +1,171 @@
 .git-blame-ignore-revs
 .gitignore
 .pre-commit-config.yaml
 AUTHORS
 CODE_OF_CONDUCT.md
 CONTRIBUTORS
 LICENSE
 MANIFEST.in
 Makefile
 Makefile.local
 README.md
 codemeta.json
 conftest.py
 mypy.ini
 pyproject.toml
 pytest.ini
 requirements-swh.txt
 requirements-test.txt
 requirements.txt
 setup.cfg
 setup.py
 tox.ini
 docs/.gitignore
 docs/Makefile
 docs/Makefile.local
 docs/README.md
 docs/cli.rst
 docs/conf.py
 docs/dev-info.rst
 docs/index.rst
 docs/metadata-workflow.rst
 docs/_static/.placeholder
 docs/_templates/.placeholder
 docs/images/.gitignore
 docs/images/Makefile
 docs/images/tasks-metadata-indexers.uml
 sql/bin/db-upgrade
 sql/bin/dot_add_content
 sql/doc/json
 sql/doc/json/.gitignore
 sql/doc/json/Makefile
 sql/doc/json/indexer_configuration.tool_configuration.schema.json
 sql/doc/json/revision_metadata.translated_metadata.json
 sql/json/.gitignore
 sql/json/Makefile
 sql/json/indexer_configuration.tool_configuration.schema.json
 sql/json/revision_metadata.translated_metadata.json
 swh/__init__.py
 swh.indexer.egg-info/PKG-INFO
 swh.indexer.egg-info/SOURCES.txt
 swh.indexer.egg-info/dependency_links.txt
 swh.indexer.egg-info/entry_points.txt
 swh.indexer.egg-info/requires.txt
 swh.indexer.egg-info/top_level.txt
 swh/indexer/__init__.py
 swh/indexer/cli.py
 swh/indexer/codemeta.py
 swh/indexer/fossology_license.py
 swh/indexer/indexer.py
 swh/indexer/journal_client.py
 swh/indexer/metadata.py
 swh/indexer/metadata_detector.py
 swh/indexer/mimetype.py
 swh/indexer/namespaces.py
 swh/indexer/origin_head.py
 swh/indexer/py.typed
 swh/indexer/rehash.py
 swh/indexer/tasks.py
 swh/indexer/data/Gitea.csv
 swh/indexer/data/composer.csv
 swh/indexer/data/nuget.csv
 swh/indexer/data/pubspec.csv
 swh/indexer/data/codemeta/CITATION
 swh/indexer/data/codemeta/LICENSE
 swh/indexer/data/codemeta/codemeta.jsonld
 swh/indexer/data/codemeta/crosswalk.csv
 swh/indexer/metadata_dictionary/__init__.py
 swh/indexer/metadata_dictionary/base.py
 swh/indexer/metadata_dictionary/cff.py
 swh/indexer/metadata_dictionary/codemeta.py
 swh/indexer/metadata_dictionary/composer.py
 swh/indexer/metadata_dictionary/dart.py
 swh/indexer/metadata_dictionary/gitea.py
 swh/indexer/metadata_dictionary/github.py
 swh/indexer/metadata_dictionary/maven.py
 swh/indexer/metadata_dictionary/npm.py
 swh/indexer/metadata_dictionary/nuget.py
 swh/indexer/metadata_dictionary/python.py
 swh/indexer/metadata_dictionary/ruby.py
 swh/indexer/metadata_dictionary/utils.py
 swh/indexer/sql/10-superuser-init.sql
 swh/indexer/sql/20-enums.sql
 swh/indexer/sql/30-schema.sql
 swh/indexer/sql/50-data.sql
 swh/indexer/sql/50-func.sql
 swh/indexer/sql/60-indexes.sql
 swh/indexer/sql/upgrades/115.sql
 swh/indexer/sql/upgrades/116.sql
 swh/indexer/sql/upgrades/117.sql
 swh/indexer/sql/upgrades/118.sql
 swh/indexer/sql/upgrades/119.sql
 swh/indexer/sql/upgrades/120.sql
 swh/indexer/sql/upgrades/121.sql
 swh/indexer/sql/upgrades/122.sql
 swh/indexer/sql/upgrades/123.sql
 swh/indexer/sql/upgrades/124.sql
 swh/indexer/sql/upgrades/125.sql
 swh/indexer/sql/upgrades/126.sql
 swh/indexer/sql/upgrades/127.sql
 swh/indexer/sql/upgrades/128.sql
 swh/indexer/sql/upgrades/129.sql
 swh/indexer/sql/upgrades/130.sql
 swh/indexer/sql/upgrades/131.sql
 swh/indexer/sql/upgrades/132.sql
 swh/indexer/sql/upgrades/133.sql
 swh/indexer/sql/upgrades/134.sql
 swh/indexer/sql/upgrades/135.sql
+swh/indexer/sql/upgrades/136.sql
+swh/indexer/sql/upgrades/137.sql
 swh/indexer/storage/__init__.py
 swh/indexer/storage/converters.py
 swh/indexer/storage/db.py
 swh/indexer/storage/exc.py
 swh/indexer/storage/in_memory.py
 swh/indexer/storage/interface.py
 swh/indexer/storage/metrics.py
 swh/indexer/storage/model.py
 swh/indexer/storage/writer.py
 swh/indexer/storage/api/__init__.py
 swh/indexer/storage/api/client.py
 swh/indexer/storage/api/serializers.py
 swh/indexer/storage/api/server.py
 swh/indexer/tests/__init__.py
 swh/indexer/tests/conftest.py
 swh/indexer/tests/tasks.py
 swh/indexer/tests/test_cli.py
 swh/indexer/tests/test_codemeta.py
 swh/indexer/tests/test_fossology_license.py
 swh/indexer/tests/test_indexer.py
 swh/indexer/tests/test_journal_client.py
 swh/indexer/tests/test_metadata.py
 swh/indexer/tests/test_mimetype.py
 swh/indexer/tests/test_origin_head.py
 swh/indexer/tests/test_origin_metadata.py
 swh/indexer/tests/utils.py
 swh/indexer/tests/metadata_dictionary/__init__.py
 swh/indexer/tests/metadata_dictionary/test_cff.py
 swh/indexer/tests/metadata_dictionary/test_codemeta.py
 swh/indexer/tests/metadata_dictionary/test_composer.py
 swh/indexer/tests/metadata_dictionary/test_dart.py
 swh/indexer/tests/metadata_dictionary/test_gitea.py
 swh/indexer/tests/metadata_dictionary/test_github.py
 swh/indexer/tests/metadata_dictionary/test_maven.py
 swh/indexer/tests/metadata_dictionary/test_npm.py
 swh/indexer/tests/metadata_dictionary/test_nuget.py
 swh/indexer/tests/metadata_dictionary/test_python.py
 swh/indexer/tests/metadata_dictionary/test_ruby.py
 swh/indexer/tests/storage/__init__.py
 swh/indexer/tests/storage/conftest.py
 swh/indexer/tests/storage/generate_data_test.py
 swh/indexer/tests/storage/test_api_client.py
 swh/indexer/tests/storage/test_converters.py
 swh/indexer/tests/storage/test_in_memory.py
 swh/indexer/tests/storage/test_init.py
 swh/indexer/tests/storage/test_metrics.py
 swh/indexer/tests/storage/test_model.py
 swh/indexer/tests/storage/test_server.py
 swh/indexer/tests/storage/test_storage.py
 swh/indexer/tests/zz_celery/README
 swh/indexer/tests/zz_celery/__init__.py
 swh/indexer/tests/zz_celery/test_tasks.py
\ No newline at end of file
diff --git a/swh/indexer/sql/20-enums.sql b/swh/indexer/sql/20-enums.sql
index a357eb5..e69de29 100644
--- a/swh/indexer/sql/20-enums.sql
+++ b/swh/indexer/sql/20-enums.sql
@@ -1,100 +0,0 @@
-create type languages as enum ( 'abap', 'abnf', 'actionscript',
-  'actionscript-3', 'ada', 'adl', 'agda', 'alloy', 'ambienttalk',
-  'antlr', 'antlr-with-actionscript-target', 'antlr-with-c#-target',
-  'antlr-with-cpp-target', 'antlr-with-java-target',
-  'antlr-with-objectivec-target', 'antlr-with-perl-target',
-  'antlr-with-python-target', 'antlr-with-ruby-target', 'apacheconf',
-  'apl', 'applescript', 'arduino', 'aspectj', 'aspx-cs', 'aspx-vb',
-  'asymptote', 'autohotkey', 'autoit', 'awk', 'base-makefile', 'bash',
-  'bash-session', 'batchfile', 'bbcode', 'bc', 'befunge',
-  'blitzbasic', 'blitzmax', 'bnf', 'boo', 'boogie', 'brainfuck',
-  'bro', 'bugs', 'c', 'c#', 'c++', 'c-objdump', 'ca65-assembler',
-  'cadl', 'camkes', 'cbm-basic-v2', 'ceylon', 'cfengine3',
-  'cfstatement', 'chaiscript', 'chapel', 'cheetah', 'cirru', 'clay',
-  'clojure', 'clojurescript', 'cmake', 'cobol', 'cobolfree',
-  'coffeescript', 'coldfusion-cfc', 'coldfusion-html', 'common-lisp',
-  'component-pascal', 'coq', 'cpp-objdump', 'cpsa', 'crmsh', 'croc',
-  'cryptol', 'csound-document', 'csound-orchestra', 'csound-score',
-  'css', 'css+django/jinja', 'css+genshi-text', 'css+lasso',
-  'css+mako', 'css+mozpreproc', 'css+myghty', 'css+php', 'css+ruby',
-  'css+smarty', 'cuda', 'cypher', 'cython', 'd', 'd-objdump',
-  'darcs-patch', 'dart', 'debian-control-file', 'debian-sourcelist',
-  'delphi', 'dg', 'diff', 'django/jinja', 'docker', 'dtd', 'duel',
-  'dylan', 'dylan-session', 'dylanlid', 'earl-grey', 'easytrieve',
-  'ebnf', 'ec', 'ecl', 'eiffel', 'elixir', 'elixir-iex-session',
-  'elm', 'emacslisp', 'embedded-ragel', 'erb', 'erlang',
-  'erlang-erl-session', 'evoque', 'ezhil', 'factor', 'fancy',
-  'fantom', 'felix', 'fish', 'fortran', 'fortranfixed', 'foxpro',
-  'fsharp', 'gap', 'gas', 'genshi', 'genshi-text', 'gettext-catalog',
-  'gherkin', 'glsl', 'gnuplot', 'go', 'golo', 'gooddata-cl', 'gosu',
-  'gosu-template', 'groff', 'groovy', 'haml', 'handlebars', 'haskell',
-  'haxe', 'hexdump', 'html', 'html+cheetah', 'html+django/jinja',
-  'html+evoque', 'html+genshi', 'html+handlebars', 'html+lasso',
-  'html+mako', 'html+myghty', 'html+php', 'html+smarty', 'html+twig',
-  'html+velocity', 'http', 'hxml', 'hy', 'hybris', 'idl', 'idris',
-  'igor', 'inform-6', 'inform-6-template', 'inform-7', 'ini', 'io',
-  'ioke', 'irc-logs', 'isabelle', 'j', 'jade', 'jags', 'jasmin',
-  'java', 'java-server-page', 'javascript', 'javascript+cheetah',
-  'javascript+django/jinja', 'javascript+genshi-text',
-  'javascript+lasso', 'javascript+mako', 'javascript+mozpreproc',
-  'javascript+myghty', 'javascript+php', 'javascript+ruby',
-  'javascript+smarty', 'jcl', 'json', 'json-ld', 'julia',
-  'julia-console', 'kal', 'kconfig', 'koka', 'kotlin', 'lasso',
-  'lean', 'lesscss', 'lighttpd-configuration-file', 'limbo', 'liquid',
-  'literate-agda', 'literate-cryptol', 'literate-haskell',
-  'literate-idris', 'livescript', 'llvm', 'logos', 'logtalk', 'lsl',
-  'lua', 'makefile', 'mako', 'maql', 'mask', 'mason', 'mathematica',
-  'matlab', 'matlab-session', 'minid', 'modelica', 'modula-2',
-  'moinmoin/trac-wiki-markup', 'monkey', 'moocode', 'moonscript',
-  'mozhashpreproc', 'mozpercentpreproc', 'mql', 'mscgen',
-  'msdos-session', 'mupad', 'mxml', 'myghty', 'mysql', 'nasm',
-  'nemerle', 'nesc', 'newlisp', 'newspeak',
-  'nginx-configuration-file', 'nimrod', 'nit', 'nix', 'nsis', 'numpy',
-  'objdump', 'objdump-nasm', 'objective-c', 'objective-c++',
-  'objective-j', 'ocaml', 'octave', 'odin', 'ooc', 'opa',
-  'openedge-abl', 'pacmanconf', 'pan', 'parasail', 'pawn', 'perl',
-  'perl6', 'php', 'pig', 'pike', 'pkgconfig', 'pl/pgsql',
-  'postgresql-console-(psql)', 'postgresql-sql-dialect', 'postscript',
-  'povray', 'powershell', 'powershell-session', 'praat', 'prolog',
-  'properties', 'protocol-buffer', 'puppet', 'pypy-log', 'python',
-  'python-3', 'python-3.0-traceback', 'python-console-session',
-  'python-traceback', 'qbasic', 'qml', 'qvto', 'racket', 'ragel',
-  'ragel-in-c-host', 'ragel-in-cpp-host', 'ragel-in-d-host',
-  'ragel-in-java-host', 'ragel-in-objective-c-host',
-  'ragel-in-ruby-host', 'raw-token-data', 'rconsole', 'rd', 'rebol',
-  'red', 'redcode', 'reg', 'resourcebundle', 'restructuredtext',
-  'rexx', 'rhtml', 'roboconf-graph', 'roboconf-instances',
-  'robotframework', 'rpmspec', 'rql', 'rsl', 'ruby',
-  'ruby-irb-session', 'rust', 's', 'sass', 'scala',
-  'scalate-server-page', 'scaml', 'scheme', 'scilab', 'scss', 'shen',
-  'slim', 'smali', 'smalltalk', 'smarty', 'snobol', 'sourcepawn',
-  'sparql', 'sql', 'sqlite3con', 'squidconf', 'stan', 'standard-ml',
-  'supercollider', 'swift', 'swig', 'systemverilog', 'tads-3', 'tap',
-  'tcl', 'tcsh', 'tcsh-session', 'tea', 'termcap', 'terminfo',
-  'terraform', 'tex', 'text-only', 'thrift', 'todotxt',
-  'trafficscript', 'treetop', 'turtle', 'twig', 'typescript',
-  'urbiscript', 'vala', 'vb.net', 'vctreestatus', 'velocity',
-  'verilog', 'vgl', 'vhdl', 'viml', 'x10', 'xml', 'xml+cheetah',
-  'xml+django/jinja', 'xml+evoque', 'xml+lasso', 'xml+mako',
-  'xml+myghty', 'xml+php', 'xml+ruby', 'xml+smarty', 'xml+velocity',
-  'xquery', 'xslt', 'xtend', 'xul+mozpreproc', 'yaml', 'yaml+jinja',
-  'zephir', 'unknown'
-);
-comment on type languages is 'Languages recognized by language indexer';
-
-create type ctags_languages as enum ( 'Ada', 'AnsiblePlaybook', 'Ant',
-  'Asm', 'Asp', 'Autoconf', 'Automake', 'Awk', 'Basic', 'BETA', 'C',
-  'C#', 'C++', 'Clojure', 'Cobol', 'CoffeeScript [disabled]', 'CSS',
-  'ctags', 'D', 'DBusIntrospect', 'Diff', 'DosBatch', 'DTS', 'Eiffel',
-  'Erlang', 'Falcon', 'Flex', 'Fortran', 'gdbinit [disabled]',
-  'Glade', 'Go', 'HTML', 'Iniconf', 'Java', 'JavaProperties',
-  'JavaScript', 'JSON', 'Lisp', 'Lua', 'M4', 'Make', 'man [disabled]',
-  'MatLab', 'Maven2', 'Myrddin', 'ObjectiveC', 'OCaml', 'OldC
-  [disabled]', 'OldC++ [disabled]', 'Pascal', 'Perl', 'Perl6', 'PHP',
-  'PlistXML', 'pod', 'Protobuf', 'Python', 'PythonLoggingConfig', 'R',
-  'RelaxNG', 'reStructuredText', 'REXX', 'RpmSpec', 'Ruby', 'Rust',
-  'Scheme', 'Sh', 'SLang', 'SML', 'SQL', 'SVG', 'SystemdUnit',
-  'SystemVerilog', 'Tcl', 'Tex', 'TTCN', 'Vera', 'Verilog', 'VHDL',
-  'Vim', 'WindRes', 'XSLT', 'YACC', 'Yaml', 'YumRepo', 'Zephir'
-);
-comment on type ctags_languages is 'Languages recognized by ctags indexer';
diff --git a/swh/indexer/sql/30-schema.sql b/swh/indexer/sql/30-schema.sql
index 08587c3..318fb69 100644
--- a/swh/indexer/sql/30-schema.sql
+++ b/swh/indexer/sql/30-schema.sql
@@ -1,148 +1,119 @@
 ---
 --- Software Heritage Indexers Data Model
 ---
 
 -- Computing metadata on sha1's contents
 
 -- a SHA1 checksum (not necessarily originating from Git)
 create domain sha1 as bytea check (length(value) = 20);
 
 -- a Git object ID, i.e., a SHA1 checksum
 create domain sha1_git as bytea check (length(value) = 20);
 
 create table indexer_configuration (
   id serial not null,
   tool_name text not null,
   tool_version text not null,
   tool_configuration jsonb
 );
 
 comment on table indexer_configuration is 'Indexer''s configuration version';
 comment on column indexer_configuration.id is 'Tool identifier';
 comment on column indexer_configuration.tool_version is 'Tool name';
 comment on column indexer_configuration.tool_version is 'Tool version';
 comment on column indexer_configuration.tool_configuration is 'Tool configuration: command line, flags, etc...';
 
 -- Properties (mimetype, encoding, etc...)
 create table content_mimetype (
   id sha1 not null,
   mimetype text not null,
   encoding text not null,
   indexer_configuration_id bigint not null
 );
 
 comment on table content_mimetype is 'Metadata associated to a raw content';
 comment on column content_mimetype.mimetype is 'Raw content Mimetype';
 comment on column content_mimetype.encoding is 'Raw content encoding';
 comment on column content_mimetype.indexer_configuration_id is 'Tool used to compute the information';
 
--- Language metadata
-create table content_language (
-  id sha1 not null,
-  lang languages not null,
-  indexer_configuration_id bigint not null
-);
-
-comment on table content_language is 'Language information on a raw content';
-comment on column content_language.lang is 'Language information';
-comment on column content_language.indexer_configuration_id is 'Tool used to compute the information';
-
--- ctags information per content
-create table content_ctags (
-  id sha1 not null,
-  name text not null,
-  kind text not null,
-  line bigint not null,
-  lang ctags_languages not null,
-  indexer_configuration_id bigint not null
-);
-
-comment on table content_ctags is 'Ctags information on a raw content';
-comment on column content_ctags.id is 'Content identifier';
-comment on column content_ctags.name is 'Symbol name';
-comment on column content_ctags.kind is 'Symbol kind (function, class, variable, const...)';
-comment on column content_ctags.line is 'Symbol line';
-comment on column content_ctags.lang is 'Language information for that content';
-comment on column content_ctags.indexer_configuration_id is 'Tool used to compute the information';
-
 create table fossology_license(
   id smallserial,
   name text not null
 );
 
 comment on table fossology_license is 'Possible license recognized by license indexer';
 comment on column fossology_license.id is 'License identifier';
 comment on column fossology_license.name is 'License name';
 
 create table content_fossology_license (
   id sha1 not null,
   license_id smallserial not null,
   indexer_configuration_id bigint not null
 );
 
 comment on table content_fossology_license is 'license associated to a raw content';
 comment on column content_fossology_license.id is 'Raw content identifier';
 comment on column content_fossology_license.license_id is 'One of the content''s license identifier';
 comment on column content_fossology_license.indexer_configuration_id is 'Tool used to compute the information';
 
 
 -- The table content_metadata provides a translation to files
 -- identified as potentially containning metadata with a translation tool (indexer_configuration_id)
 create table content_metadata(
   id                       sha1   not null,
   metadata                 jsonb  not null,
   indexer_configuration_id bigint not null
 );
 
 comment on table content_metadata is 'metadata semantically translated from a content file';
 comment on column content_metadata.id is 'sha1 of content file';
 comment on column content_metadata.metadata is 'result of translation with defined format';
 comment on column content_metadata.indexer_configuration_id is 'tool used for translation';
 
 -- The table directory_intrinsic_metadata provides a minimal set of intrinsic
 -- metadata detected with the detection  tool (indexer_configuration_id) and
 -- aggregated from the content_metadata translation.
 create table directory_intrinsic_metadata(
   id                       sha1_git   not null,
   metadata                 jsonb      not null,
   indexer_configuration_id bigint     not null,
   mappings                 text array not null
 );
 
 comment on table directory_intrinsic_metadata is 'metadata semantically detected and translated in a directory';
 comment on column directory_intrinsic_metadata.id is 'sha1_git of directory';
 comment on column directory_intrinsic_metadata.metadata is 'result of detection and translation with defined format';
 comment on column directory_intrinsic_metadata.indexer_configuration_id is 'tool used for detection';
 comment on column directory_intrinsic_metadata.mappings is 'type of metadata files used to obtain this metadata (eg. pkg-info, npm)';
 
 create table origin_intrinsic_metadata(
   id                        text       not null,  -- origin url
   metadata                  jsonb,
   indexer_configuration_id  bigint     not null,
   from_directory             sha1_git   not null,
   metadata_tsvector         tsvector,
   mappings                  text array not null
 );
 
 comment on table origin_intrinsic_metadata is 'keeps intrinsic metadata for an origin';
 comment on column origin_intrinsic_metadata.id is 'url of the origin';
 comment on column origin_intrinsic_metadata.metadata is 'metadata extracted from a directory';
 comment on column origin_intrinsic_metadata.indexer_configuration_id is 'tool used to generate this metadata';
 comment on column origin_intrinsic_metadata.from_directory is 'sha1 of the directory this metadata was copied from.';
 comment on column origin_intrinsic_metadata.mappings is 'type of metadata files used to obtain this metadata (eg. pkg-info, npm)';
 
 create table origin_extrinsic_metadata(
   id                        text       not null,  -- origin url
   metadata                  jsonb,
   indexer_configuration_id  bigint     not null,
   from_remd_id              sha1_git   not null,
   metadata_tsvector         tsvector,
   mappings                  text array not null
 );
 
 comment on table origin_extrinsic_metadata is 'keeps extrinsic metadata for an origin';
 comment on column origin_extrinsic_metadata.id is 'url of the origin';
 comment on column origin_extrinsic_metadata.metadata is 'metadata extracted from a directory';
 comment on column origin_extrinsic_metadata.indexer_configuration_id is 'tool used to generate this metadata';
 comment on column origin_extrinsic_metadata.from_remd_id is 'sha1 of the directory this metadata was copied from.';
 comment on column origin_extrinsic_metadata.mappings is 'type of metadata files used to obtain this metadata (eg. github, gitlab)';
diff --git a/swh/indexer/sql/50-func.sql b/swh/indexer/sql/50-func.sql
index d459a4a..85f292c 100644
--- a/swh/indexer/sql/50-func.sql
+++ b/swh/indexer/sql/50-func.sql
@@ -1,487 +1,382 @@
 -- Postgresql index helper function
 create or replace function hash_sha1(text)
     returns text
     language sql strict immutable
 as $$
     select encode(public.digest($1, 'sha1'), 'hex')
 $$;
 
 comment on function hash_sha1(text) is 'Compute sha1 hash as text';
 
 -- create a temporary table called tmp_TBLNAME, mimicking existing table
 -- TBLNAME
 --
 -- Args:
 --     tblname: name of the table to mimic
 create or replace function swh_mktemp(tblname regclass)
     returns void
     language plpgsql
 as $$
 begin
     execute format('
 	create temporary table if not exists tmp_%1$I
 	    (like %1$I including defaults)
 	    on commit delete rows;
       alter table tmp_%1$I drop column if exists object_id;
 	', tblname);
     return;
 end
 $$;
 
 -- create a temporary table for content_mimetype tmp_content_mimetype,
 create or replace function swh_mktemp_content_mimetype()
     returns void
     language sql
 as $$
   create temporary table if not exists tmp_content_mimetype (
     like content_mimetype including defaults
   ) on commit delete rows;
 $$;
 
 comment on function swh_mktemp_content_mimetype() IS 'Helper table to add mimetype information';
 
 -- add tmp_content_mimetype entries to content_mimetype, overwriting duplicates.
 --
 -- If filtering duplicates is in order, the call to
 -- swh_content_mimetype_missing must take place before calling this
 -- function.
 --
 -- operates in bulk: 0. swh_mktemp(content_mimetype), 1. COPY to tmp_content_mimetype,
 -- 2. call this function
 create or replace function swh_content_mimetype_add()
     returns bigint
     language plpgsql
 as $$
 declare
   res bigint;
 begin
     insert into content_mimetype (id, mimetype, encoding, indexer_configuration_id)
     select id, mimetype, encoding, indexer_configuration_id
     from tmp_content_mimetype tcm
+    order by id, indexer_configuration_id
     on conflict(id, indexer_configuration_id)
     do update set mimetype = excluded.mimetype,
                   encoding = excluded.encoding;
 
     get diagnostics res = ROW_COUNT;
     return res;
 end
 $$;
 
 comment on function swh_content_mimetype_add() IS 'Add new content mimetypes';
 
--- add tmp_content_language entries to content_language, overwriting duplicates.
---
--- If filtering duplicates is in order, the call to
--- swh_content_language_missing must take place before calling this
--- function.
---
--- operates in bulk: 0. swh_mktemp(content_language), 1. COPY to
--- tmp_content_language, 2. call this function
-create or replace function swh_content_language_add()
-    returns bigint
-    language plpgsql
-as $$
-declare
-  res bigint;
-begin
-    insert into content_language (id, lang, indexer_configuration_id)
-    select id, lang, indexer_configuration_id
-    from tmp_content_language tcl
-    on conflict(id, indexer_configuration_id)
-    do update set lang = excluded.lang;
-
-    get diagnostics res = ROW_COUNT;
-    return res;
-end
-$$;
-
-comment on function swh_content_language_add() IS 'Add new content languages';
-
--- create a temporary table for retrieving content_language
-create or replace function swh_mktemp_content_language()
-    returns void
-    language sql
-as $$
-  create temporary table if not exists tmp_content_language (
-    like content_language including defaults
-  ) on commit delete rows;
-$$;
-
-comment on function swh_mktemp_content_language() is 'Helper table to add content language';
-
-
--- create a temporary table for content_ctags tmp_content_ctags,
-create or replace function swh_mktemp_content_ctags()
-    returns void
-    language sql
-as $$
-  create temporary table if not exists tmp_content_ctags (
-    like content_ctags including defaults
-  ) on commit delete rows;
-$$;
-
-comment on function swh_mktemp_content_ctags() is 'Helper table to add content ctags';
-
-
--- add tmp_content_ctags entries to content_ctags, overwriting duplicates
---
--- operates in bulk: 0. swh_mktemp(content_ctags), 1. COPY to tmp_content_ctags,
--- 2. call this function
-create or replace function swh_content_ctags_add()
-    returns bigint
-    language plpgsql
-as $$
-declare
-  res bigint;
-begin
-    insert into content_ctags (id, name, kind, line, lang, indexer_configuration_id)
-    select id, name, kind, line, lang, indexer_configuration_id
-    from tmp_content_ctags tct
-    on conflict(id, hash_sha1(name), kind, line, lang, indexer_configuration_id)
-    do nothing;
-
-    get diagnostics res = ROW_COUNT;
-    return res;
-end
-$$;
-
-comment on function swh_content_ctags_add() IS 'Add new ctags symbols per content';
-
-create type content_ctags_signature as (
-  id sha1,
-  name text,
-  kind text,
-  line bigint,
-  lang ctags_languages,
-  tool_id integer,
-  tool_name text,
-  tool_version text,
-  tool_configuration jsonb
-);
-
--- Search within ctags content.
---
-create or replace function swh_content_ctags_search(
-       expression text,
-       l integer default 10,
-       last_sha1 sha1 default '\x0000000000000000000000000000000000000000')
-    returns setof content_ctags_signature
-    language sql
-as $$
-    select c.id, name, kind, line, lang,
-           i.id as tool_id, tool_name, tool_version, tool_configuration
-    from content_ctags c
-    inner join indexer_configuration i on i.id = c.indexer_configuration_id
-    where hash_sha1(name) = hash_sha1(expression)
-    and c.id > last_sha1
-    order by id
-    limit l;
-$$;
-
-comment on function swh_content_ctags_search(text, integer, sha1) IS 'Equality search through ctags'' symbols';
-
-
 -- create a temporary table for content_fossology_license tmp_content_fossology_license,
 create or replace function swh_mktemp_content_fossology_license()
     returns void
     language sql
 as $$
   create temporary table if not exists tmp_content_fossology_license (
     id                       sha1,
     license                  text,
     indexer_configuration_id integer
   ) on commit delete rows;
 $$;
 
 comment on function swh_mktemp_content_fossology_license() is 'Helper table to add content license';
 
 -- add tmp_content_fossology_license entries to content_fossology_license,
 -- overwriting duplicates.
 --
 -- operates in bulk: 0. swh_mktemp(content_fossology_license), 1. COPY to
 -- tmp_content_fossology_license, 2. call this function
 create or replace function swh_content_fossology_license_add()
   returns bigint
   language plpgsql
 as $$
 declare
   res bigint;
 begin
     -- insert unknown licenses first
     insert into fossology_license (name)
     select distinct license from tmp_content_fossology_license tmp
     where not exists (select 1 from fossology_license where name=tmp.license)
     on conflict(name) do nothing;
 
     insert into content_fossology_license (id, license_id, indexer_configuration_id)
     select tcl.id,
           (select id from fossology_license where name = tcl.license) as license,
           indexer_configuration_id
     from tmp_content_fossology_license tcl
+    order by tcl.id, license, indexer_configuration_id
     on conflict(id, license_id, indexer_configuration_id)
     do update set license_id = excluded.license_id;
 
     get diagnostics res = ROW_COUNT;
     return res;
 end
 $$;
 
 comment on function swh_content_fossology_license_add() IS 'Add new content licenses';
 
 
 -- content_metadata functions
 
 -- add tmp_content_metadata entries to content_metadata, overwriting duplicates
 --
 -- If filtering duplicates is in order, the call to
 -- swh_content_metadata_missing must take place before calling this
 -- function.
 --
--- operates in bulk: 0. swh_mktemp(content_language), 1. COPY to
+-- operates in bulk: 0. swh_mktemp(content_metadata), 1. COPY to
 -- tmp_content_metadata, 2. call this function
 create or replace function swh_content_metadata_add()
     returns bigint
     language plpgsql
 as $$
 declare
   res bigint;
 begin
     insert into content_metadata (id, metadata, indexer_configuration_id)
     select id, metadata, indexer_configuration_id
     from tmp_content_metadata tcm
+    order by id, indexer_configuration_id
     on conflict(id, indexer_configuration_id)
     do update set metadata = excluded.metadata;
 
     get diagnostics res = ROW_COUNT;
     return res;
 end
 $$;
 
 comment on function swh_content_metadata_add() IS 'Add new content metadata';
 
 -- create a temporary table for retrieving content_metadata
 create or replace function swh_mktemp_content_metadata()
     returns void
     language sql
 as $$
   create temporary table if not exists tmp_content_metadata (
     like content_metadata including defaults
   ) on commit delete rows;
 $$;
 
 comment on function swh_mktemp_content_metadata() is 'Helper table to add content metadata';
 
 -- end content_metadata functions
 
 -- add tmp_directory_intrinsic_metadata entries to directory_intrinsic_metadata,
 -- overwriting duplicates.
 --
 -- If filtering duplicates is in order, the call to
 -- swh_directory_intrinsic_metadata_missing must take place before calling this
 -- function.
 --
--- operates in bulk: 0. swh_mktemp(content_language), 1. COPY to
+-- operates in bulk: 0. swh_mktemp(directory_intrinsic_metadata), 1. COPY to
 -- tmp_directory_intrinsic_metadata, 2. call this function
 create or replace function swh_directory_intrinsic_metadata_add()
     returns bigint
     language plpgsql
 as $$
 declare
   res bigint;
 begin
     insert into directory_intrinsic_metadata (id, metadata, mappings, indexer_configuration_id)
     select id, metadata, mappings, indexer_configuration_id
     from tmp_directory_intrinsic_metadata tcm
+    order by id, indexer_configuration_id
     on conflict(id, indexer_configuration_id)
     do update set
         metadata = excluded.metadata,
         mappings = excluded.mappings;
 
     get diagnostics res = ROW_COUNT;
     return res;
 end
 $$;
 
 comment on function swh_directory_intrinsic_metadata_add() IS 'Add new directory intrinsic metadata';
 
 -- create a temporary table for retrieving directory_intrinsic_metadata
 create or replace function swh_mktemp_directory_intrinsic_metadata()
     returns void
     language sql
 as $$
   create temporary table if not exists tmp_directory_intrinsic_metadata (
     like directory_intrinsic_metadata including defaults
   ) on commit delete rows;
 $$;
 
 comment on function swh_mktemp_directory_intrinsic_metadata() is 'Helper table to add directory intrinsic metadata';
 
 -- create a temporary table for retrieving origin_intrinsic_metadata
 create or replace function swh_mktemp_origin_intrinsic_metadata()
     returns void
     language sql
 as $$
   create temporary table if not exists tmp_origin_intrinsic_metadata (
     like origin_intrinsic_metadata including defaults
   ) on commit delete rows;
 $$;
 
 comment on function swh_mktemp_origin_intrinsic_metadata() is 'Helper table to add origin intrinsic metadata';
 
 create or replace function swh_mktemp_indexer_configuration()
     returns void
     language sql
 as $$
     create temporary table if not exists tmp_indexer_configuration (
       like indexer_configuration including defaults
     ) on commit delete rows;
     alter table tmp_indexer_configuration drop column if exists id;
 $$;
 
 -- add tmp_origin_intrinsic_metadata entries to origin_intrinsic_metadata,
 -- overwriting duplicates.
 --
 -- If filtering duplicates is in order, the call to
 -- swh_origin_intrinsic_metadata_missing must take place before calling this
 -- function.
 --
--- operates in bulk: 0. swh_mktemp(content_language), 1. COPY to
+-- operates in bulk: 0. swh_mktemp(origin_intrinsic_metadata), 1. COPY to
 -- tmp_origin_intrinsic_metadata, 2. call this function
 create or replace function swh_origin_intrinsic_metadata_add()
     returns bigint
     language plpgsql
 as $$
 declare
   res bigint;
 begin
     perform swh_origin_intrinsic_metadata_compute_tsvector();
 
     insert into origin_intrinsic_metadata (id, metadata, indexer_configuration_id, from_directory, metadata_tsvector, mappings)
     select id, metadata, indexer_configuration_id, from_directory,
            metadata_tsvector, mappings
     from tmp_origin_intrinsic_metadata
+    order by id, indexer_configuration_id
     on conflict(id, indexer_configuration_id)
     do update set
         metadata = excluded.metadata,
         metadata_tsvector = excluded.metadata_tsvector,
         mappings = excluded.mappings,
         from_directory = excluded.from_directory;
 
     get diagnostics res = ROW_COUNT;
     return res;
 end
 $$;
 
 comment on function swh_origin_intrinsic_metadata_add() IS 'Add new origin intrinsic metadata';
 
 
 -- Compute the metadata_tsvector column in tmp_origin_intrinsic_metadata.
 --
 -- It uses the "pg_catalog.simple" dictionary, as it has no stopword,
 -- so it should be suitable for proper names and non-English text.
 create or replace function swh_origin_intrinsic_metadata_compute_tsvector()
     returns void
     language plpgsql
 as $$
 begin
     update tmp_origin_intrinsic_metadata
         set metadata_tsvector = to_tsvector('pg_catalog.simple', metadata);
 end
 $$;
 
 -- create a temporary table for retrieving origin_extrinsic_metadata
 create or replace function swh_mktemp_origin_extrinsic_metadata()
     returns void
     language sql
 as $$
   create temporary table if not exists tmp_origin_extrinsic_metadata (
     like origin_extrinsic_metadata including defaults
   ) on commit delete rows;
 $$;
 
 comment on function swh_mktemp_origin_extrinsic_metadata() is 'Helper table to add origin extrinsic metadata';
 
 create or replace function swh_mktemp_indexer_configuration()
     returns void
     language sql
 as $$
     create temporary table if not exists tmp_indexer_configuration (
       like indexer_configuration including defaults
     ) on commit delete rows;
     alter table tmp_indexer_configuration drop column if exists id;
 $$;
 
 -- add tmp_origin_extrinsic_metadata entries to origin_extrinsic_metadata,
 -- overwriting duplicates.
 --
 -- If filtering duplicates is in order, the call to
 -- swh_origin_extrinsic_metadata_missing must take place before calling this
 -- function.
 --
--- operates in bulk: 0. swh_mktemp(content_language), 1. COPY to
+-- operates in bulk: 0. swh_mktemp(origin_extrinsic_metadata), 1. COPY to
 -- tmp_origin_extrinsic_metadata, 2. call this function
 create or replace function swh_origin_extrinsic_metadata_add()
     returns bigint
     language plpgsql
 as $$
 declare
   res bigint;
 begin
     perform swh_origin_extrinsic_metadata_compute_tsvector();
 
     insert into origin_extrinsic_metadata (id, metadata, indexer_configuration_id, from_remd_id, metadata_tsvector, mappings)
     select id, metadata, indexer_configuration_id, from_remd_id,
            metadata_tsvector, mappings
     from tmp_origin_extrinsic_metadata
+    order by id, indexer_configuration_id
     on conflict(id, indexer_configuration_id)
     do update set
         metadata = excluded.metadata,
         metadata_tsvector = excluded.metadata_tsvector,
         mappings = excluded.mappings,
         from_remd_id = excluded.from_remd_id;
 
     get diagnostics res = ROW_COUNT;
     return res;
 end
 $$;
 
 comment on function swh_origin_extrinsic_metadata_add() IS 'Add new origin extrinsic metadata';
 
 
 -- Compute the metadata_tsvector column in tmp_origin_extrinsic_metadata.
 --
 -- It uses the "pg_catalog.simple" dictionary, as it has no stopword,
 -- so it should be suitable for proper names and non-English text.
 create or replace function swh_origin_extrinsic_metadata_compute_tsvector()
     returns void
     language plpgsql
 as $$
 begin
     update tmp_origin_extrinsic_metadata
         set metadata_tsvector = to_tsvector('pg_catalog.simple', metadata);
 end
 $$;
 
 
 -- add tmp_indexer_configuration entries to indexer_configuration,
 -- overwriting duplicates if any.
 --
 -- operates in bulk: 0. create temporary tmp_indexer_configuration, 1. COPY to
 -- it, 2. call this function to insert and filtering out duplicates
 create or replace function swh_indexer_configuration_add()
     returns setof indexer_configuration
     language plpgsql
 as $$
 begin
       insert into indexer_configuration(tool_name, tool_version, tool_configuration)
       select tool_name, tool_version, tool_configuration from tmp_indexer_configuration tmp
+      order by tool_name, tool_version, tool_configuration
       on conflict(tool_name, tool_version, tool_configuration) do nothing;
 
       return query
           select id, tool_name, tool_version, tool_configuration
           from tmp_indexer_configuration join indexer_configuration
               using(tool_name, tool_version, tool_configuration);
 
       return;
 end
 $$;
diff --git a/swh/indexer/sql/60-indexes.sql b/swh/indexer/sql/60-indexes.sql
index 5b42af7..20fe3ca 100644
--- a/swh/indexer/sql/60-indexes.sql
+++ b/swh/indexer/sql/60-indexes.sql
@@ -1,79 +1,64 @@
 -- fossology_license
 create unique index fossology_license_pkey on fossology_license(id);
 alter table fossology_license add primary key using index fossology_license_pkey;
 
 create unique index on fossology_license(name);
 
 -- indexer_configuration
 create unique index concurrently indexer_configuration_pkey on indexer_configuration(id);
 alter table indexer_configuration add primary key using index indexer_configuration_pkey;
 
 create unique index on indexer_configuration(tool_name, tool_version, tool_configuration);
 
--- content_ctags
-create index on content_ctags(id);
-create index on content_ctags(hash_sha1(name));
-create unique index on content_ctags(id, hash_sha1(name), kind, line, lang, indexer_configuration_id);
-
-alter table content_ctags add constraint content_ctags_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
-alter table content_ctags validate constraint content_ctags_indexer_configuration_id_fkey;
-
 -- content_metadata
 create unique index content_metadata_pkey on content_metadata(id, indexer_configuration_id);
 alter table content_metadata add primary key using index content_metadata_pkey;
 
 alter table content_metadata add constraint content_metadata_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
 alter table content_metadata validate constraint content_metadata_indexer_configuration_id_fkey;
 
 -- directory_intrinsic_metadata
 create unique index directory_intrinsic_metadata_pkey on directory_intrinsic_metadata(id, indexer_configuration_id);
 alter table directory_intrinsic_metadata add primary key using index directory_intrinsic_metadata_pkey;
 
 alter table directory_intrinsic_metadata add constraint directory_intrinsic_metadata_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
 alter table directory_intrinsic_metadata validate constraint directory_intrinsic_metadata_indexer_configuration_id_fkey;
 
 -- content_mimetype
 create unique index content_mimetype_pkey on content_mimetype(id, indexer_configuration_id);
 alter table content_mimetype add primary key using index content_mimetype_pkey;
 
 alter table content_mimetype add constraint content_mimetype_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
 alter table content_mimetype validate constraint content_mimetype_indexer_configuration_id_fkey;
 
 create index on content_mimetype(id) where mimetype like 'text/%';
 
--- content_language
-create unique index content_language_pkey on content_language(id, indexer_configuration_id);
-alter table content_language add primary key using index content_language_pkey;
-
-alter table content_language add constraint content_language_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
-alter table content_language validate constraint content_language_indexer_configuration_id_fkey;
-
 -- content_fossology_license
 create unique index content_fossology_license_pkey on content_fossology_license(id, license_id, indexer_configuration_id);
 alter table content_fossology_license add primary key using index content_fossology_license_pkey;
 
 alter table content_fossology_license add constraint content_fossology_license_license_id_fkey foreign key (license_id) references fossology_license(id) not valid;
 alter table content_fossology_license validate constraint content_fossology_license_license_id_fkey;
 
 alter table content_fossology_license add constraint content_fossology_license_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
 alter table content_fossology_license validate constraint content_fossology_license_indexer_configuration_id_fkey;
 
 -- origin_intrinsic_metadata
 create unique index origin_intrinsic_metadata_pkey on origin_intrinsic_metadata(id, indexer_configuration_id);
 alter table origin_intrinsic_metadata add primary key using index origin_intrinsic_metadata_pkey;
 
 alter table origin_intrinsic_metadata add constraint origin_intrinsic_metadata_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
 alter table origin_intrinsic_metadata validate constraint origin_intrinsic_metadata_indexer_configuration_id_fkey;
 
 create index origin_intrinsic_metadata_fulltext_idx on origin_intrinsic_metadata using gin (metadata_tsvector);
 create index origin_intrinsic_metadata_mappings_idx on origin_intrinsic_metadata using gin (mappings);
 
 -- origin_extrinsic_metadata
 create unique index origin_extrinsic_metadata_pkey on origin_extrinsic_metadata(id, indexer_configuration_id);
 alter table origin_extrinsic_metadata add primary key using index origin_extrinsic_metadata_pkey;
 
 alter table origin_extrinsic_metadata add constraint origin_extrinsic_metadata_indexer_configuration_id_fkey foreign key (indexer_configuration_id) references indexer_configuration(id) not valid;
 alter table origin_extrinsic_metadata validate constraint origin_extrinsic_metadata_indexer_configuration_id_fkey;
 
 create index origin_extrinsic_metadata_fulltext_idx on origin_extrinsic_metadata using gin (metadata_tsvector);
 create index origin_extrinsic_metadata_mappings_idx on origin_extrinsic_metadata using gin (mappings);
diff --git a/swh/indexer/sql/upgrades/136.sql b/swh/indexer/sql/upgrades/136.sql
new file mode 100644
index 0000000..01499ac
--- /dev/null
+++ b/swh/indexer/sql/upgrades/136.sql
@@ -0,0 +1,214 @@
+-- SWH Indexer DB schema upgrade
+-- from_version: 135
+-- to_version: 136
+-- description: Insert from temporary tables in consistent order
+
+insert into dbversion(version, release, description)
+      values(136, now(), 'Work In Progress');
+
+
+create or replace function swh_content_mimetype_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    insert into content_mimetype (id, mimetype, encoding, indexer_configuration_id)
+    select id, mimetype, encoding, indexer_configuration_id
+    from tmp_content_mimetype tcm
+    order by id, indexer_configuration_id
+    on conflict(id, indexer_configuration_id)
+    do update set mimetype = excluded.mimetype,
+                  encoding = excluded.encoding;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_content_language_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    insert into content_language (id, lang, indexer_configuration_id)
+    select id, lang, indexer_configuration_id
+    from tmp_content_language tcl
+    order by id, indexer_configuration_id
+    on conflict(id, indexer_configuration_id)
+    do update set lang = excluded.lang;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_content_ctags_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    insert into content_ctags (id, name, kind, line, lang, indexer_configuration_id)
+    select id, name, kind, line, lang, indexer_configuration_id
+    from tmp_content_ctags tct
+    order by id, hash_sha1(name), kind, line, lang, indexer_configuration_id
+    on conflict(id, hash_sha1(name), kind, line, lang, indexer_configuration_id)
+    do nothing;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_content_fossology_license_add()
+  returns bigint
+  language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    -- insert unknown licenses first
+    insert into fossology_license (name)
+    select distinct license from tmp_content_fossology_license tmp
+    where not exists (select 1 from fossology_license where name=tmp.license)
+    on conflict(name) do nothing;
+
+    insert into content_fossology_license (id, license_id, indexer_configuration_id)
+    select tcl.id,
+          (select id from fossology_license where name = tcl.license) as license,
+          indexer_configuration_id
+    from tmp_content_fossology_license tcl
+    order by tcl.id, license, indexer_configuration_id
+    on conflict(id, license_id, indexer_configuration_id)
+    do update set license_id = excluded.license_id;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_content_metadata_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    insert into content_metadata (id, metadata, indexer_configuration_id)
+    select id, metadata, indexer_configuration_id
+    from tmp_content_metadata tcm
+    order by id, indexer_configuration_id
+    on conflict(id, indexer_configuration_id)
+    do update set metadata = excluded.metadata;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_directory_intrinsic_metadata_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    insert into directory_intrinsic_metadata (id, metadata, mappings, indexer_configuration_id)
+    select id, metadata, mappings, indexer_configuration_id
+    from tmp_directory_intrinsic_metadata tcm
+    order by id, indexer_configuration_id
+    on conflict(id, indexer_configuration_id)
+    do update set
+        metadata = excluded.metadata,
+        mappings = excluded.mappings;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_origin_intrinsic_metadata_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    perform swh_origin_intrinsic_metadata_compute_tsvector();
+
+    insert into origin_intrinsic_metadata (id, metadata, indexer_configuration_id, from_directory, metadata_tsvector, mappings)
+    select id, metadata, indexer_configuration_id, from_directory,
+           metadata_tsvector, mappings
+    from tmp_origin_intrinsic_metadata
+    order by id, indexer_configuration_id
+    on conflict(id, indexer_configuration_id)
+    do update set
+        metadata = excluded.metadata,
+        metadata_tsvector = excluded.metadata_tsvector,
+        mappings = excluded.mappings,
+        from_directory = excluded.from_directory;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_origin_extrinsic_metadata_add()
+    returns bigint
+    language plpgsql
+as $$
+declare
+  res bigint;
+begin
+    perform swh_origin_extrinsic_metadata_compute_tsvector();
+
+    insert into origin_extrinsic_metadata (id, metadata, indexer_configuration_id, from_remd_id, metadata_tsvector, mappings)
+    select id, metadata, indexer_configuration_id, from_remd_id,
+           metadata_tsvector, mappings
+    from tmp_origin_extrinsic_metadata
+    order by id, indexer_configuration_id
+    on conflict(id, indexer_configuration_id)
+    do update set
+        metadata = excluded.metadata,
+        metadata_tsvector = excluded.metadata_tsvector,
+        mappings = excluded.mappings,
+        from_remd_id = excluded.from_remd_id;
+
+    get diagnostics res = ROW_COUNT;
+    return res;
+end
+$$;
+
+
+create or replace function swh_indexer_configuration_add()
+    returns setof indexer_configuration
+    language plpgsql
+as $$
+begin
+      insert into indexer_configuration(tool_name, tool_version, tool_configuration)
+      select tool_name, tool_version, tool_configuration from tmp_indexer_configuration tmp
+      order by tool_name, tool_version, tool_configuration
+      on conflict(tool_name, tool_version, tool_configuration) do nothing;
+
+      return query
+          select id, tool_name, tool_version, tool_configuration
+          from tmp_indexer_configuration join indexer_configuration
+              using(tool_name, tool_version, tool_configuration);
+
+      return;
+end
+$$;
+
+
diff --git a/swh/indexer/sql/upgrades/137.sql b/swh/indexer/sql/upgrades/137.sql
new file mode 100644
index 0000000..a7d69f1
--- /dev/null
+++ b/swh/indexer/sql/upgrades/137.sql
@@ -0,0 +1,23 @@
+-- SWH Indexer DB schema upgrade
+-- from_version: 136
+-- to_version: 137
+-- description: Drop content_language and content_ctags tables and related functions
+
+insert into dbversion(version, release, description)
+      values(137, now(), 'Work In Progress');
+
+drop function swh_content_language_add;
+drop function swh_mktemp_content_language();
+drop function swh_mktemp_content_ctags();
+drop function swh_content_ctags_add();
+drop function swh_content_ctags_search;
+
+drop index content_language_pkey;
+
+drop table content_language;
+drop table content_ctags;
+
+drop type languages;
+drop type ctags_languages;
+drop type content_ctags_signature;
+
diff --git a/swh/indexer/storage/__init__.py b/swh/indexer/storage/__init__.py
index 2c7bbc2..2d0d7d7 100644
--- a/swh/indexer/storage/__init__.py
+++ b/swh/indexer/storage/__init__.py
@@ -1,722 +1,716 @@
 # Copyright (C) 2015-2022  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 from collections import Counter
 from importlib import import_module
 import json
 from typing import Dict, Iterable, List, Optional, Tuple, Union
 import warnings
 
 import psycopg2
 import psycopg2.pool
 
 from swh.core.db.common import db_transaction
 from swh.indexer.storage.interface import IndexerStorageInterface
 from swh.model.hashutil import hash_to_bytes, hash_to_hex
 from swh.model.model import SHA1_SIZE
 from swh.storage.exc import StorageDBError
 from swh.storage.utils import get_partition_bounds_bytes
 
 from . import converters
 from .db import Db
 from .exc import DuplicateId, IndexerStorageArgumentException
 from .interface import PagedResult, Sha1
 from .metrics import process_metrics, send_metric, timed
 from .model import (
     ContentLicenseRow,
     ContentMetadataRow,
     ContentMimetypeRow,
     DirectoryIntrinsicMetadataRow,
     OriginExtrinsicMetadataRow,
     OriginIntrinsicMetadataRow,
 )
 from .writer import JournalWriter
 
 INDEXER_CFG_KEY = "indexer_storage"
 
 
 MAPPING_NAMES = ["cff", "codemeta", "gemspec", "maven", "npm", "pkg-info"]
 
 
 SERVER_IMPLEMENTATIONS: Dict[str, str] = {
     "postgresql": ".IndexerStorage",
     "remote": ".api.client.RemoteStorage",
     "memory": ".in_memory.IndexerStorage",
     # deprecated
     "local": ".IndexerStorage",
 }
 
 
 def sanitize_json(doc):
     """Recursively replaces NUL characters, as postgresql does not allow
     them in text fields."""
     if isinstance(doc, str):
         return doc.replace("\x00", "")
     elif not hasattr(doc, "__iter__"):
         return doc
     elif isinstance(doc, dict):
         return {sanitize_json(k): sanitize_json(v) for (k, v) in doc.items()}
     elif isinstance(doc, (list, tuple)):
         return [sanitize_json(v) for v in doc]
     else:
         raise TypeError(f"Unexpected object type in sanitize_json: {doc}")
 
 
 def get_indexer_storage(cls: str, **kwargs) -> IndexerStorageInterface:
     """Instantiate an indexer storage implementation of class `cls` with arguments
     `kwargs`.
 
     Args:
         cls: indexer storage class (local, remote or memory)
         kwargs: dictionary of arguments passed to the
             indexer storage class constructor
 
     Returns:
         an instance of swh.indexer.storage
 
     Raises:
         ValueError if passed an unknown storage class.
 
     """
     if "args" in kwargs:
         warnings.warn(
             'Explicit "args" key is deprecated, use keys directly instead.',
             DeprecationWarning,
         )
         kwargs = kwargs["args"]
 
     class_path = SERVER_IMPLEMENTATIONS.get(cls)
     if class_path is None:
         raise ValueError(
             f"Unknown indexer storage class `{cls}`. "
             f"Supported: {', '.join(SERVER_IMPLEMENTATIONS)}"
         )
 
     (module_path, class_name) = class_path.rsplit(".", 1)
     module = import_module(module_path if module_path else ".", package=__package__)
     BackendClass = getattr(module, class_name)
     check_config = kwargs.pop("check_config", {})
     idx_storage = BackendClass(**kwargs)
     if check_config:
         if not idx_storage.check_config(**check_config):
             raise EnvironmentError("Indexer storage check config failed")
     return idx_storage
 
 
 def check_id_duplicates(data):
     """
     If any two row models in `data` have the same unique key, raises
     a `ValueError`.
 
     Values associated to the key must be hashable.
 
     Args:
         data (List[dict]): List of dictionaries to be inserted
 
     >>> check_id_duplicates([
     ...     ContentLicenseRow(id=b'foo', indexer_configuration_id=42, license="GPL"),
     ...     ContentLicenseRow(id=b'foo', indexer_configuration_id=32, license="GPL"),
     ... ])
     >>> check_id_duplicates([
     ...     ContentLicenseRow(id=b'foo', indexer_configuration_id=42, license="AGPL"),
     ...     ContentLicenseRow(id=b'foo', indexer_configuration_id=42, license="AGPL"),
     ... ])
     Traceback (most recent call last):
     ...
     swh.indexer.storage.exc.DuplicateId: [{'id': b'foo', 'indexer_configuration_id': 42, 'license': 'AGPL'}]
 
     """  # noqa
     counter = Counter(tuple(sorted(item.unique_key().items())) for item in data)
     duplicates = [id_ for (id_, count) in counter.items() if count >= 2]
     if duplicates:
         raise DuplicateId(list(map(dict, duplicates)))
 
 
 class IndexerStorage:
     """SWH Indexer Storage Datastore"""
 
-    current_version = 135
+    current_version = 137
 
     def __init__(self, db, min_pool_conns=1, max_pool_conns=10, journal_writer=None):
         """
         Args:
             db: either a libpq connection string, or a psycopg2 connection
             journal_writer: configuration passed to
                             `swh.journal.writer.get_journal_writer`
 
         """
         self.journal_writer = JournalWriter(self._tool_get_from_id, journal_writer)
         try:
             if isinstance(db, psycopg2.extensions.connection):
                 self._pool = None
                 self._db = Db(db)
             else:
                 self._pool = psycopg2.pool.ThreadedConnectionPool(
                     min_pool_conns, max_pool_conns, db
                 )
                 self._db = None
         except psycopg2.OperationalError as e:
             raise StorageDBError(e)
 
     def get_db(self):
         if self._db:
             return self._db
         return Db.from_pool(self._pool)
 
     def put_db(self, db):
         if db is not self._db:
             db.put_conn()
 
     @timed
     @db_transaction()
     def check_config(self, *, check_write, db=None, cur=None):
         # Check permissions on one of the tables
         if check_write:
             check = "INSERT"
         else:
             check = "SELECT"
 
         cur.execute(
             "select has_table_privilege(current_user, 'content_mimetype', %s)",  # noqa
             (check,),
         )
         return cur.fetchone()[0]
 
     @timed
     @db_transaction()
     def content_mimetype_missing(
         self, mimetypes: Iterable[Dict], db=None, cur=None
     ) -> List[Tuple[Sha1, int]]:
         return [obj[0] for obj in db.content_mimetype_missing_from_list(mimetypes, cur)]
 
     @timed
     @db_transaction()
     def get_partition(
         self,
         indexer_type: str,
         indexer_configuration_id: int,
         partition_id: int,
         nb_partitions: int,
         page_token: Optional[str] = None,
         limit: int = 1000,
         with_textual_data=False,
         db=None,
         cur=None,
     ) -> PagedResult[Sha1]:
         """Retrieve ids of content with `indexer_type` within within partition partition_id
         bound by limit.
 
         Args:
             **indexer_type**: Type of data content to index (mimetype, etc...)
             **indexer_configuration_id**: The tool used to index data
             **partition_id**: index of the partition to fetch
             **nb_partitions**: total number of partitions to split into
             **page_token**: opaque token used for pagination
             **limit**: Limit result (default to 1000)
             **with_textual_data** (bool): Deal with only textual content (True) or all
                 content (all contents by defaults, False)
 
         Raises:
             IndexerStorageArgumentException for;
             - limit to None
             - wrong indexer_type provided
 
         Returns:
             PagedResult of Sha1. If next_page_token is None, there is no more data to
             fetch
 
         """
         if limit is None:
             raise IndexerStorageArgumentException("limit should not be None")
         if indexer_type not in db.content_indexer_names:
             err = f"Wrong type. Should be one of [{','.join(db.content_indexer_names)}]"
             raise IndexerStorageArgumentException(err)
 
         start, end = get_partition_bounds_bytes(partition_id, nb_partitions, SHA1_SIZE)
         if page_token is not None:
             start = hash_to_bytes(page_token)
         if end is None:
             end = b"\xff" * SHA1_SIZE
 
         next_page_token: Optional[str] = None
         ids = [
             row[0]
             for row in db.content_get_range(
                 indexer_type,
                 start,
                 end,
                 indexer_configuration_id,
                 limit=limit + 1,
                 with_textual_data=with_textual_data,
                 cur=cur,
             )
         ]
 
         if len(ids) >= limit:
             next_page_token = hash_to_hex(ids[-1])
             ids = ids[:limit]
 
         assert len(ids) <= limit
         return PagedResult(results=ids, next_page_token=next_page_token)
 
     @timed
     @db_transaction()
     def content_mimetype_get_partition(
         self,
         indexer_configuration_id: int,
         partition_id: int,
         nb_partitions: int,
         page_token: Optional[str] = None,
         limit: int = 1000,
         db=None,
         cur=None,
     ) -> PagedResult[Sha1]:
         return self.get_partition(
             "mimetype",
             indexer_configuration_id,
             partition_id,
             nb_partitions,
             page_token=page_token,
             limit=limit,
             db=db,
             cur=cur,
         )
 
     @timed
     @process_metrics
     @db_transaction()
     def content_mimetype_add(
         self,
         mimetypes: List[ContentMimetypeRow],
         db=None,
         cur=None,
     ) -> Dict[str, int]:
         check_id_duplicates(mimetypes)
-        mimetypes.sort(key=lambda m: m.id)
         self.journal_writer.write_additions("content_mimetype", mimetypes)
         db.mktemp_content_mimetype(cur)
         db.copy_to(
             [m.to_dict() for m in mimetypes],
             "tmp_content_mimetype",
             ["id", "mimetype", "encoding", "indexer_configuration_id"],
             cur,
         )
         count = db.content_mimetype_add_from_temp(cur)
         return {"content_mimetype:add": count}
 
     @timed
     @db_transaction()
     def content_mimetype_get(
         self, ids: Iterable[Sha1], db=None, cur=None
     ) -> List[ContentMimetypeRow]:
         return [
             ContentMimetypeRow.from_dict(
                 converters.db_to_mimetype(dict(zip(db.content_mimetype_cols, c)))
             )
             for c in db.content_mimetype_get_from_list(ids, cur)
         ]
 
     @timed
     @db_transaction()
     def content_fossology_license_get(
         self, ids: Iterable[Sha1], db=None, cur=None
     ) -> List[ContentLicenseRow]:
         return [
             ContentLicenseRow.from_dict(
                 converters.db_to_fossology_license(
                     dict(zip(db.content_fossology_license_cols, c))
                 )
             )
             for c in db.content_fossology_license_get_from_list(ids, cur)
         ]
 
     @timed
     @process_metrics
     @db_transaction()
     def content_fossology_license_add(
         self,
         licenses: List[ContentLicenseRow],
         db=None,
         cur=None,
     ) -> Dict[str, int]:
         check_id_duplicates(licenses)
-        licenses.sort(key=lambda m: m.id)
         self.journal_writer.write_additions("content_fossology_license", licenses)
         db.mktemp_content_fossology_license(cur)
         db.copy_to(
             [license.to_dict() for license in licenses],
             tblname="tmp_content_fossology_license",
             columns=["id", "license", "indexer_configuration_id"],
             cur=cur,
         )
         count = db.content_fossology_license_add_from_temp(cur)
         return {"content_fossology_license:add": count}
 
     @timed
     @db_transaction()
     def content_fossology_license_get_partition(
         self,
         indexer_configuration_id: int,
         partition_id: int,
         nb_partitions: int,
         page_token: Optional[str] = None,
         limit: int = 1000,
         db=None,
         cur=None,
     ) -> PagedResult[Sha1]:
         return self.get_partition(
             "fossology_license",
             indexer_configuration_id,
             partition_id,
             nb_partitions,
             page_token=page_token,
             limit=limit,
             with_textual_data=True,
             db=db,
             cur=cur,
         )
 
     @timed
     @db_transaction()
     def content_metadata_missing(
         self, metadata: Iterable[Dict], db=None, cur=None
     ) -> List[Tuple[Sha1, int]]:
         return [obj[0] for obj in db.content_metadata_missing_from_list(metadata, cur)]
 
     @timed
     @db_transaction()
     def content_metadata_get(
         self, ids: Iterable[Sha1], db=None, cur=None
     ) -> List[ContentMetadataRow]:
         return [
             ContentMetadataRow.from_dict(
                 converters.db_to_metadata(dict(zip(db.content_metadata_cols, c)))
             )
             for c in db.content_metadata_get_from_list(ids, cur)
         ]
 
     @timed
     @process_metrics
     @db_transaction()
     def content_metadata_add(
         self,
         metadata: List[ContentMetadataRow],
         db=None,
         cur=None,
     ) -> Dict[str, int]:
         check_id_duplicates(metadata)
-        metadata.sort(key=lambda m: m.id)
         self.journal_writer.write_additions("content_metadata", metadata)
 
         db.mktemp_content_metadata(cur)
 
         rows = [m.to_dict() for m in metadata]
         for row in rows:
             row["metadata"] = sanitize_json(row["metadata"])
 
         db.copy_to(
             rows,
             "tmp_content_metadata",
             ["id", "metadata", "indexer_configuration_id"],
             cur,
         )
         count = db.content_metadata_add_from_temp(cur)
         return {
             "content_metadata:add": count,
         }
 
     @timed
     @db_transaction()
     def directory_intrinsic_metadata_missing(
         self, metadata: Iterable[Dict], db=None, cur=None
     ) -> List[Tuple[Sha1, int]]:
         return [
             obj[0]
             for obj in db.directory_intrinsic_metadata_missing_from_list(metadata, cur)
         ]
 
     @timed
     @db_transaction()
     def directory_intrinsic_metadata_get(
         self, ids: Iterable[Sha1], db=None, cur=None
     ) -> List[DirectoryIntrinsicMetadataRow]:
         return [
             DirectoryIntrinsicMetadataRow.from_dict(
                 converters.db_to_metadata(
                     dict(zip(db.directory_intrinsic_metadata_cols, c))
                 )
             )
             for c in db.directory_intrinsic_metadata_get_from_list(ids, cur)
         ]
 
     @timed
     @process_metrics
     @db_transaction()
     def directory_intrinsic_metadata_add(
         self,
         metadata: List[DirectoryIntrinsicMetadataRow],
         db=None,
         cur=None,
     ) -> Dict[str, int]:
         check_id_duplicates(metadata)
-        metadata.sort(key=lambda m: m.id)
         self.journal_writer.write_additions("directory_intrinsic_metadata", metadata)
 
         db.mktemp_directory_intrinsic_metadata(cur)
 
         rows = [m.to_dict() for m in metadata]
         for row in rows:
             row["metadata"] = sanitize_json(row["metadata"])
 
         db.copy_to(
             rows,
             "tmp_directory_intrinsic_metadata",
             ["id", "metadata", "mappings", "indexer_configuration_id"],
             cur,
         )
         count = db.directory_intrinsic_metadata_add_from_temp(cur)
         return {
             "directory_intrinsic_metadata:add": count,
         }
 
     @timed
     @db_transaction()
     def origin_intrinsic_metadata_get(
         self, urls: Iterable[str], db=None, cur=None
     ) -> List[OriginIntrinsicMetadataRow]:
         return [
             OriginIntrinsicMetadataRow.from_dict(
                 converters.db_to_metadata(
                     dict(zip(db.origin_intrinsic_metadata_cols, c))
                 )
             )
             for c in db.origin_intrinsic_metadata_get_from_list(urls, cur)
         ]
 
     @timed
     @process_metrics
     @db_transaction()
     def origin_intrinsic_metadata_add(
         self,
         metadata: List[OriginIntrinsicMetadataRow],
         db=None,
         cur=None,
     ) -> Dict[str, int]:
         check_id_duplicates(metadata)
-        metadata.sort(key=lambda m: m.id)
         self.journal_writer.write_additions("origin_intrinsic_metadata", metadata)
 
         db.mktemp_origin_intrinsic_metadata(cur)
 
         rows = [m.to_dict() for m in metadata]
         for row in rows:
             row["metadata"] = sanitize_json(row["metadata"])
 
         db.copy_to(
             rows,
             "tmp_origin_intrinsic_metadata",
             [
                 "id",
                 "metadata",
                 "indexer_configuration_id",
                 "from_directory",
                 "mappings",
             ],
             cur,
         )
         count = db.origin_intrinsic_metadata_add_from_temp(cur)
         return {
             "origin_intrinsic_metadata:add": count,
         }
 
     @timed
     @db_transaction()
     def origin_intrinsic_metadata_search_fulltext(
         self, conjunction: List[str], limit: int = 100, db=None, cur=None
     ) -> List[OriginIntrinsicMetadataRow]:
         return [
             OriginIntrinsicMetadataRow.from_dict(
                 converters.db_to_metadata(
                     dict(zip(db.origin_intrinsic_metadata_cols, c))
                 )
             )
             for c in db.origin_intrinsic_metadata_search_fulltext(
                 conjunction, limit=limit, cur=cur
             )
         ]
 
     @timed
     @db_transaction()
     def origin_intrinsic_metadata_search_by_producer(
         self,
         page_token: str = "",
         limit: int = 100,
         ids_only: bool = False,
         mappings: Optional[List[str]] = None,
         tool_ids: Optional[List[int]] = None,
         db=None,
         cur=None,
     ) -> PagedResult[Union[str, OriginIntrinsicMetadataRow]]:
         assert isinstance(page_token, str)
         # we go to limit+1 to check whether we should add next_page_token in
         # the response
         rows = db.origin_intrinsic_metadata_search_by_producer(
             page_token, limit + 1, ids_only, mappings, tool_ids, cur
         )
         next_page_token = None
         if ids_only:
             results = [origin for (origin,) in rows]
             if len(results) > limit:
                 results[limit:] = []
                 next_page_token = results[-1]
         else:
             results = [
                 OriginIntrinsicMetadataRow.from_dict(
                     converters.db_to_metadata(
                         dict(zip(db.origin_intrinsic_metadata_cols, row))
                     )
                 )
                 for row in rows
             ]
             if len(results) > limit:
                 results[limit:] = []
                 next_page_token = results[-1].id
 
         return PagedResult(
             results=results,
             next_page_token=next_page_token,
         )
 
     @timed
     @db_transaction()
     def origin_intrinsic_metadata_stats(self, db=None, cur=None):
         mapping_names = [m for m in MAPPING_NAMES]
         select_parts = []
 
         # Count rows for each mapping
         for mapping_name in mapping_names:
             select_parts.append(
                 (
                     "sum(case when (mappings @> ARRAY['%s']) "
                     "         then 1 else 0 end)"
                 )
                 % mapping_name
             )
 
         # Total
         select_parts.append("sum(1)")
 
         # Rows whose metadata has at least one key that is not '@context'
         select_parts.append(
             "sum(case when ('{}'::jsonb @> (metadata - '@context')) "
             "         then 0 else 1 end)"
         )
         cur.execute(
             "select " + ", ".join(select_parts) + " from origin_intrinsic_metadata"
         )
         results = dict(zip(mapping_names + ["total", "non_empty"], cur.fetchone()))
         return {
             "total": results.pop("total"),
             "non_empty": results.pop("non_empty"),
             "per_mapping": results,
         }
 
     @timed
     @db_transaction()
     def origin_extrinsic_metadata_get(
         self, urls: Iterable[str], db=None, cur=None
     ) -> List[OriginExtrinsicMetadataRow]:
         return [
             OriginExtrinsicMetadataRow.from_dict(
                 converters.db_to_metadata(
                     dict(zip(db.origin_extrinsic_metadata_cols, c))
                 )
             )
             for c in db.origin_extrinsic_metadata_get_from_list(urls, cur)
         ]
 
     @timed
     @process_metrics
     @db_transaction()
     def origin_extrinsic_metadata_add(
         self,
         metadata: List[OriginExtrinsicMetadataRow],
         db=None,
         cur=None,
     ) -> Dict[str, int]:
         check_id_duplicates(metadata)
-        metadata.sort(key=lambda m: m.id)
         self.journal_writer.write_additions("origin_extrinsic_metadata", metadata)
 
         db.mktemp_origin_extrinsic_metadata(cur)
 
         rows = [m.to_dict() for m in metadata]
         for row in rows:
             row["metadata"] = sanitize_json(row["metadata"])
 
         db.copy_to(
             rows,
             "tmp_origin_extrinsic_metadata",
             [
                 "id",
                 "metadata",
                 "indexer_configuration_id",
                 "from_remd_id",
                 "mappings",
             ],
             cur,
         )
         count = db.origin_extrinsic_metadata_add_from_temp(cur)
         return {
             "origin_extrinsic_metadata:add": count,
         }
 
     @timed
     @db_transaction()
     def indexer_configuration_add(self, tools, db=None, cur=None):
         db.mktemp_indexer_configuration(cur)
         db.copy_to(
             tools,
             "tmp_indexer_configuration",
             ["tool_name", "tool_version", "tool_configuration"],
             cur,
         )
 
         tools = db.indexer_configuration_add_from_temp(cur)
         results = [dict(zip(db.indexer_configuration_cols, line)) for line in tools]
         send_metric(
             "indexer_configuration:add",
             len(results),
             method_name="indexer_configuration_add",
         )
         return results
 
     @timed
     @db_transaction()
     def indexer_configuration_get(self, tool, db=None, cur=None):
         tool_conf = tool["tool_configuration"]
         if isinstance(tool_conf, dict):
             tool_conf = json.dumps(tool_conf)
         idx = db.indexer_configuration_get(
             tool["tool_name"], tool["tool_version"], tool_conf
         )
         if not idx:
             return None
         return dict(zip(db.indexer_configuration_cols, idx))
 
     @db_transaction()
     def _tool_get_from_id(self, id_, db, cur):
         tool = dict(
             zip(
                 db.indexer_configuration_cols,
                 db.indexer_configuration_get_from_id(id_, cur),
             )
         )
         return {
             "id": tool["id"],
             "name": tool["tool_name"],
             "version": tool["tool_version"],
             "configuration": tool["tool_configuration"],
         }