diff --git a/PKG-INFO b/PKG-INFO
index 558c278..ee55f34 100644
--- a/PKG-INFO
+++ b/PKG-INFO
@@ -1,70 +1,70 @@
 Metadata-Version: 2.1
 Name: swh.journal
-Version: 0.0.21
+Version: 0.0.23
 Summary: Software Heritage Journal utilities
 Home-page: https://forge.softwareheritage.org/diffusion/DJNL/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 License: UNKNOWN
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-journal
 Description: swh-journal
         ===========
         
         Persistent logger of changes to the archive, with publish-subscribe support.
         
         See the
         [documentation](https://docs.softwareheritage.org/devel/swh-journal/index.html#software-heritage-journal)
         for more details.
         
         # Local test
         
         As a pre-requisite, you need a kakfa installation path.
         The following target will take care of this:
         
         ```
         make install
         ```
         
         Then, provided you are in the right virtual environment as described
         in the [swh getting-started](https://docs.softwareheritage.org/devel/developer-setup.html#developer-setup):
         
         ```
         pytest
         ```
         
         or:
         
         ```
         tox
         ```
         
         
         # Running
         
         ## publisher
         
         Command:
         ```
         $ swh-journal --config-file ~/.config/swh/journal/publisher.yml \
                       publisher
         ```
         
         # Auto-completion
         
         To have the completion, add the following in your
         ~/.virtualenvs/swh/bin/postactivate:
         
         ```
         eval "$(_SWH_JOURNAL_COMPLETE=$autocomplete_cmd swh-journal)"
         ```
         
 Platform: UNKNOWN
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Description-Content-Type: text/markdown
 Provides-Extra: testing
diff --git a/swh.journal.egg-info/PKG-INFO b/swh.journal.egg-info/PKG-INFO
index 558c278..ee55f34 100644
--- a/swh.journal.egg-info/PKG-INFO
+++ b/swh.journal.egg-info/PKG-INFO
@@ -1,70 +1,70 @@
 Metadata-Version: 2.1
 Name: swh.journal
-Version: 0.0.21
+Version: 0.0.23
 Summary: Software Heritage Journal utilities
 Home-page: https://forge.softwareheritage.org/diffusion/DJNL/
 Author: Software Heritage developers
 Author-email: swh-devel@inria.fr
 License: UNKNOWN
 Project-URL: Bug Reports, https://forge.softwareheritage.org/maniphest
 Project-URL: Funding, https://www.softwareheritage.org/donate
 Project-URL: Source, https://forge.softwareheritage.org/source/swh-journal
 Description: swh-journal
         ===========
         
         Persistent logger of changes to the archive, with publish-subscribe support.
         
         See the
         [documentation](https://docs.softwareheritage.org/devel/swh-journal/index.html#software-heritage-journal)
         for more details.
         
         # Local test
         
         As a pre-requisite, you need a kakfa installation path.
         The following target will take care of this:
         
         ```
         make install
         ```
         
         Then, provided you are in the right virtual environment as described
         in the [swh getting-started](https://docs.softwareheritage.org/devel/developer-setup.html#developer-setup):
         
         ```
         pytest
         ```
         
         or:
         
         ```
         tox
         ```
         
         
         # Running
         
         ## publisher
         
         Command:
         ```
         $ swh-journal --config-file ~/.config/swh/journal/publisher.yml \
                       publisher
         ```
         
         # Auto-completion
         
         To have the completion, add the following in your
         ~/.virtualenvs/swh/bin/postactivate:
         
         ```
         eval "$(_SWH_JOURNAL_COMPLETE=$autocomplete_cmd swh-journal)"
         ```
         
 Platform: UNKNOWN
 Classifier: Programming Language :: Python :: 3
 Classifier: Intended Audience :: Developers
 Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
 Classifier: Operating System :: OS Independent
 Classifier: Development Status :: 5 - Production/Stable
 Description-Content-Type: text/markdown
 Provides-Extra: testing
diff --git a/swh/journal/cli.py b/swh/journal/cli.py
index 7c40adf..13455d6 100644
--- a/swh/journal/cli.py
+++ b/swh/journal/cli.py
@@ -1,236 +1,263 @@
 # Copyright (C) 2016-2019 The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import functools
 import logging
 import mmap
 import os
 import time
 
 import click
 
+try:
+    from systemd.daemon import notify
+except ImportError:
+    notify = None
+
 from swh.core import config
 from swh.core.cli import CONTEXT_SETTINGS
 from swh.model.model import SHA1_SIZE
 from swh.storage import get_storage
 from swh.objstorage import get_objstorage
 
 from swh.journal.client import JournalClient
 from swh.journal.replay import is_hash_in_bytearray
 from swh.journal.replay import process_replay_objects
 from swh.journal.replay import process_replay_objects_content
 from swh.journal.backfill import JournalBackfiller
 
 
 @click.group(name='journal', context_settings=CONTEXT_SETTINGS)
 @click.option('--config-file', '-C', default=None,
               type=click.Path(exists=True, dir_okay=False,),
               help="Configuration file.")
 @click.pass_context
 def cli(ctx, config_file):
     """Software Heritage Journal tools.
 
     The journal is a persistent logger of changes to the archive, with
     publish-subscribe support.
 
     """
     if not config_file:
         config_file = os.environ.get('SWH_CONFIG_FILENAME')
 
     if config_file:
         if not os.path.exists(config_file):
             raise ValueError('%s does not exist' % config_file)
         conf = config.read(config_file)
     else:
         conf = {}
 
     ctx.ensure_object(dict)
 
     ctx.obj['config'] = conf
 
 
 def get_journal_client(ctx, **kwargs):
     conf = ctx.obj['config'].get('journal', {})
     conf.update({k: v for (k, v) in kwargs.items() if v not in (None, ())})
     if not conf.get('brokers'):
         ctx.fail('You must specify at least one kafka broker.')
     if not isinstance(conf['brokers'], (list, tuple)):
         conf['brokers'] = [conf['brokers']]
     return JournalClient(**conf)
 
 
 @cli.command()
 @click.option('--max-messages', '-m', default=None, type=int,
               help='Maximum number of objects to replay. Default is to '
                    'run forever.')
 @click.option('--broker', 'brokers', type=str, multiple=True,
               help='Kafka broker to connect to. '
                    '(deprecated, use the config file instead)')
 @click.option('--prefix', type=str, default=None,
               help='Prefix of Kafka topic names to read from. '
                    '(deprecated, use the config file instead)')
 @click.option('--group-id', type=str,
               help='Name of the group id for reading from Kafka. '
                    '(deprecated, use the config file instead)')
 @click.pass_context
 def replay(ctx, brokers, prefix, group_id, max_messages):
     """Fill a Storage by reading a Journal.
 
     There can be several 'replayers' filling a Storage as long as they use
     the same `group-id`.
     """
     logger = logging.getLogger(__name__)
     conf = ctx.obj['config']
     try:
         storage = get_storage(**conf.pop('storage'))
     except KeyError:
         ctx.fail('You must have a storage configured in your config file.')
 
     client = get_journal_client(
         ctx, brokers=brokers, prefix=prefix, group_id=group_id,
         max_messages=max_messages)
     worker_fn = functools.partial(process_replay_objects, storage=storage)
 
+    if notify:
+        notify('READY=1')
+
     try:
         nb_messages = 0
         last_log_time = 0
         while not max_messages or nb_messages < max_messages:
             nb_messages += client.process(worker_fn)
+            if notify:
+                notify('WATCHDOG=1')
             if time.time() - last_log_time >= 60:
                 # Log at most once per minute.
                 logger.info('Processed %d messages.' % nb_messages)
                 last_log_time = time.time()
     except KeyboardInterrupt:
         ctx.exit(0)
     else:
         print('Done.')
     finally:
+        if notify:
+            notify('STOPPING=1')
         client.close()
 
 
 @cli.command()
 @click.argument('object_type')
 @click.option('--start-object', default=None)
 @click.option('--end-object', default=None)
 @click.option('--dry-run', is_flag=True, default=False)
 @click.pass_context
 def backfiller(ctx, object_type, start_object, end_object, dry_run):
     """Run the backfiller
 
     The backfiller list objects from a Storage and produce journal entries from
     there.
 
     Typically used to rebuild a journal or compensate for missing objects in a
     journal (eg. due to a downtime of this later).
 
     The configuration file requires the following entries:
     - brokers: a list of kafka endpoints (the journal) in which entries will be
                added.
     - storage_dbconn: URL to connect to the storage DB.
     - prefix: the prefix of the topics (topics will be <prefix>.<object_type>).
     - client_id: the kafka client ID.
 
     """
     conf = ctx.obj['config']
     backfiller = JournalBackfiller(conf)
+
+    if notify:
+        notify('READY=1')
+
     try:
         backfiller.run(
             object_type=object_type,
             start_object=start_object, end_object=end_object,
             dry_run=dry_run)
     except KeyboardInterrupt:
+        if notify:
+            notify('STOPPING=1')
         ctx.exit(0)
 
 
 @cli.command('content-replay')
 @click.option('--max-messages', '-m', default=None, type=int,
               help='Maximum number of objects to replay. Default is to '
                    'run forever.')
 @click.option('--broker', 'brokers', type=str, multiple=True,
               help='Kafka broker to connect to.'
                    '(deprecated, use the config file instead)')
 @click.option('--prefix', type=str, default=None,
               help='Prefix of Kafka topic names to read from.'
                    '(deprecated, use the config file instead)')
 @click.option('--group-id', type=str,
               help='Name of the group id for reading from Kafka.'
                    '(deprecated, use the config file instead)')
 @click.option('--exclude-sha1-file', default=None, type=click.File('rb'),
               help='File containing a sorted array of hashes to be excluded.')
 @click.pass_context
 def content_replay(ctx, max_messages,
                    brokers, prefix, group_id, exclude_sha1_file):
     """Fill a destination Object Storage (typically a mirror) by reading a Journal
     and retrieving objects from an existing source ObjStorage.
 
     There can be several 'replayers' filling a given ObjStorage as long as they
     use the same `group-id`.
 
     This service retrieves object ids to copy from the 'content' topic. It will
     only copy object's content if the object's description in the kafka
     nmessage has the status:visible set.
 
     `--exclude-sha1-file` may be used to exclude some hashes to speed-up the
     replay in case many of the contents are already in the destination
     objstorage. It must contain a concatenation of all (sha1) hashes,
     and it must be sorted.
     This file will not be fully loaded into memory at any given time,
     so it can be arbitrarily large.
     """
     logger = logging.getLogger(__name__)
     conf = ctx.obj['config']
     try:
         objstorage_src = get_objstorage(**conf.pop('objstorage_src'))
     except KeyError:
         ctx.fail('You must have a source objstorage configured in '
                  'your config file.')
     try:
         objstorage_dst = get_objstorage(**conf.pop('objstorage_dst'))
     except KeyError:
         ctx.fail('You must have a destination objstorage configured '
                  'in your config file.')
 
     if exclude_sha1_file:
         map_ = mmap.mmap(exclude_sha1_file.fileno(), 0, prot=mmap.PROT_READ)
         if map_.size() % SHA1_SIZE != 0:
             ctx.fail('--exclude-sha1 must link to a file whose size is an '
                      'exact multiple of %d bytes.' % SHA1_SIZE)
         nb_excluded_hashes = int(map_.size()/SHA1_SIZE)
 
         def exclude_fn(obj):
             return is_hash_in_bytearray(obj['sha1'], map_, nb_excluded_hashes)
     else:
         exclude_fn = None
 
     client = get_journal_client(
         ctx, brokers=brokers, prefix=prefix, group_id=group_id,
         max_messages=max_messages, object_types=('content',))
     worker_fn = functools.partial(process_replay_objects_content,
                                   src=objstorage_src,
                                   dst=objstorage_dst,
                                   exclude_fn=exclude_fn)
 
+    if notify:
+        notify('READY=1')
+
     try:
         nb_messages = 0
         last_log_time = 0
         while not max_messages or nb_messages < max_messages:
             nb_messages += client.process(worker_fn)
+            if notify:
+                notify('WATCHDOG=1')
             if time.time() - last_log_time >= 60:
                 # Log at most once per minute.
                 logger.info('Processed %d messages.' % nb_messages)
                 last_log_time = time.time()
     except KeyboardInterrupt:
         ctx.exit(0)
     else:
         print('Done.')
+    finally:
+        if notify:
+            notify('STOPPING=1')
+        client.close()
 
 
 def main():
     logging.basicConfig()
     return cli(auto_envvar_prefix='SWH_JOURNAL')
 
 
 if __name__ == '__main__':
     main()
diff --git a/swh/journal/client.py b/swh/journal/client.py
index fb8310f..0d481b0 100644
--- a/swh/journal/client.py
+++ b/swh/journal/client.py
@@ -1,187 +1,194 @@
 # Copyright (C) 2017  The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 from collections import defaultdict
 import logging
 import time
 
-from confluent_kafka import Consumer, KafkaException
+from confluent_kafka import Consumer, KafkaException, KafkaError
 
 from .serializers import kafka_to_value
 from swh.journal import DEFAULT_PREFIX
 
 logger = logging.getLogger(__name__)
 rdkafka_logger = logging.getLogger(__name__ + '.rdkafka')
 
 
 # Only accepted offset reset policy accepted
 ACCEPTED_OFFSET_RESET = ['earliest', 'latest']
 
 # Only accepted object types
 ACCEPTED_OBJECT_TYPES = [
     'content',
     'directory',
     'revision',
     'release',
     'snapshot',
     'origin',
     'origin_visit'
 ]
 
+# Errors that Kafka raises too often and are not useful; therefore they
+# we lower their log level to DEBUG instead of INFO.
+_SPAMMY_ERRORS = [
+    KafkaError._NO_OFFSET,
+]
+
 
 def _error_cb(error):
     if error.fatal():
         raise KafkaException(error)
-    logger.info('Received non-fatal kafka error: %s', error)
+    if error.code() in _SPAMMY_ERRORS:
+        logger.debug('Received non-fatal kafka error: %s', error)
+    else:
+        logger.info('Received non-fatal kafka error: %s', error)
 
 
 def _on_commit(error, partitions):
     if error is not None:
         _error_cb(error)
 
 
 class JournalClient:
     """A base client for the Software Heritage journal.
 
     The current implementation of the journal uses Apache Kafka
     brokers to publish messages under a given topic prefix, with each
     object type using a specific topic under that prefix. If the 'prefix'
     argument is None (default value), it will take the default value
     'swh.journal.objects'.
 
     Clients subscribe to events specific to each object type as listed in the
     `object_types` argument (if unset, defaults to all accepted object types).
 
     Clients can be sharded by setting the `group_id` to a common
     value across instances. The journal will share the message
     throughput across the nodes sharing the same group_id.
 
     Messages are processed by the `worker_fn` callback passed to the
     `process` method, in batches of maximum `max_messages`.
 
     Any other named argument is passed directly to KafkaConsumer().
 
     """
     def __init__(
             self, brokers, group_id, prefix=None, object_types=None,
             max_messages=0, process_timeout=0, auto_offset_reset='earliest',
             **kwargs):
         if prefix is None:
             prefix = DEFAULT_PREFIX
         if object_types is None:
             object_types = ACCEPTED_OBJECT_TYPES
         if auto_offset_reset not in ACCEPTED_OFFSET_RESET:
             raise ValueError(
                 'Option \'auto_offset_reset\' only accept %s, not %s' %
                 (ACCEPTED_OFFSET_RESET, auto_offset_reset))
 
         for object_type in object_types:
             if object_type not in ACCEPTED_OBJECT_TYPES:
                 raise ValueError(
                     'Option \'object_types\' only accepts %s, not %s.' %
                     (ACCEPTED_OBJECT_TYPES, object_type))
 
         self.value_deserializer = kafka_to_value
 
         if isinstance(brokers, str):
             brokers = [brokers]
 
         debug_logging = rdkafka_logger.isEnabledFor(logging.DEBUG)
         if debug_logging and 'debug' not in kwargs:
             kwargs['debug'] = 'consumer'
 
         consumer_settings = {
             **kwargs,
             'bootstrap.servers': ','.join(brokers),
             'auto.offset.reset': auto_offset_reset,
             'group.id': group_id,
             'on_commit': _on_commit,
             'error_cb': _error_cb,
             'enable.auto.commit': False,
             'logger': rdkafka_logger,
         }
 
         logger.debug('Consumer settings: %s', consumer_settings)
 
         self.consumer = Consumer(consumer_settings)
 
         topics = ['%s.%s' % (prefix, object_type)
                   for object_type in object_types]
 
         logger.debug('Upstream topics: %s',
                      self.consumer.list_topics(timeout=10))
         logger.debug('Subscribing to: %s', topics)
 
         self.consumer.subscribe(topics=topics)
 
         self.max_messages = max_messages
         self.process_timeout = process_timeout
 
         self._object_types = object_types
 
     def process(self, worker_fn):
         """Polls Kafka for a batch of messages, and calls the worker_fn
         with these messages.
 
         Args:
             worker_fn Callable[Dict[str, List[dict]]]: Function called with
                                                        the messages as
                                                        argument.
         """
         start_time = time.monotonic()
         nb_messages = 0
 
         objects = defaultdict(list)
 
         while True:
             # timeout for message poll
             timeout = 1.0
 
             elapsed = time.monotonic() - start_time
             if self.process_timeout:
                 if elapsed + 0.01 >= self.process_timeout:
                     break
 
                 timeout = self.process_timeout - elapsed
 
             num_messages = 20
 
             if self.max_messages:
                 if nb_messages >= self.max_messages:
                     break
                 num_messages = min(num_messages, self.max_messages-nb_messages)
 
             messages = self.consumer.consume(
                 timeout=timeout, num_messages=num_messages)
             if not messages:
                 continue
 
             for message in messages:
                 error = message.error()
                 if error is not None:
-                    if error.fatal():
-                        raise KafkaException(error)
-                    logger.info('Received non-fatal kafka error: %s', error)
+                    _error_cb(error)
                     continue
 
                 nb_messages += 1
 
                 object_type = message.topic().split('.')[-1]
                 # Got a message from a topic we did not subscribe to.
                 assert object_type in self._object_types, object_type
 
                 objects[object_type].append(
                     self.value_deserializer(message.value())
                 )
 
             if objects:
                 worker_fn(dict(objects))
                 objects.clear()
 
                 self.consumer.commit()
         return nb_messages
 
     def close(self):
         self.consumer.close()
diff --git a/swh/journal/replay.py b/swh/journal/replay.py
index cd0357c..85b253e 100644
--- a/swh/journal/replay.py
+++ b/swh/journal/replay.py
@@ -1,406 +1,416 @@
 # Copyright (C) 2019 The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import copy
 from time import time
 import logging
 from contextlib import contextmanager
 
+try:
+    from systemd.daemon import notify
+except ImportError:
+    notify = None
+
 from swh.core.statsd import statsd
 from swh.model.identifiers import normalize_timestamp
 from swh.model.hashutil import hash_to_hex
 from swh.model.model import SHA1_SIZE
 from swh.objstorage.objstorage import ID_HASH_ALGO
 from swh.storage import HashCollision
 
 logger = logging.getLogger(__name__)
 
 GRAPH_OPERATIONS_METRIC = "swh_graph_replayer_operations_total"
 GRAPH_DURATION_METRIC = "swh_graph_replayer_duration_seconds"
 CONTENT_OPERATIONS_METRIC = "swh_content_replayer_operations_total"
 CONTENT_BYTES_METRIC = "swh_content_replayer_bytes"
 CONTENT_DURATION_METRIC = "swh_content_replayer_duration_seconds"
 
 
 def process_replay_objects(all_objects, *, storage):
     for (object_type, objects) in all_objects.items():
         logger.debug("Inserting %s %s objects", len(objects), object_type)
         with statsd.timed(GRAPH_DURATION_METRIC,
                           tags={'object_type': object_type}):
             _insert_objects(object_type, objects, storage)
         statsd.increment(GRAPH_OPERATIONS_METRIC, len(objects),
                          tags={'object_type': object_type})
+    if notify:
+        notify('WATCHDOG=1')
 
 
 def _fix_revision_pypi_empty_string(rev):
     """PyPI loader failed to encode empty strings as bytes, see:
     swh:1:rev:8f0095ee0664867055d03de9bcc8f95b91d8a2b9
     or https://forge.softwareheritage.org/D1772
     """
     rev = {
         **rev,
         'author': rev['author'].copy(),
         'committer': rev['committer'].copy(),
     }
     if rev['author'].get('email') == '':
         rev['author']['email'] = b''
     if rev['author'].get('name') == '':
         rev['author']['name'] = b''
     if rev['committer'].get('email') == '':
         rev['committer']['email'] = b''
     if rev['committer'].get('name') == '':
         rev['committer']['name'] = b''
     return rev
 
 
 def _fix_revision_transplant_source(rev):
     if rev.get('metadata') and rev['metadata'].get('extra_headers'):
         rev = copy.deepcopy(rev)
         rev['metadata']['extra_headers'] = [
             [key, value.encode('ascii')]
             if key == 'transplant_source' and isinstance(value, str)
             else [key, value]
             for (key, value) in rev['metadata']['extra_headers']]
     return rev
 
 
 def _check_date(date):
     """Returns whether the date can be represented in backends with sane
     limits on timestamps and timezeones (resp. signed 64-bits and
     signed 16 bits), and that microseconds is valid (ie. between 0 and 10^6).
     """
     date = normalize_timestamp(date)
     return (-2**63 <= date['timestamp']['seconds'] < 2**63) \
         and (0 <= date['timestamp']['microseconds'] < 10**6) \
         and (-2**15 <= date['offset'] < 2**15)
 
 
 def _check_revision_date(rev):
     """Exclude revisions with invalid dates.
     See https://forge.softwareheritage.org/T1339"""
     return _check_date(rev['date']) and _check_date(rev['committer_date'])
 
 
 def _fix_revisions(revisions):
     good_revisions = []
     for rev in revisions:
         rev = _fix_revision_pypi_empty_string(rev)
         rev = _fix_revision_transplant_source(rev)
         if not _check_revision_date(rev):
             logging.warning('Excluding revision (invalid date): %r', rev)
             continue
         if rev not in good_revisions:
             good_revisions.append(rev)
     return good_revisions
 
 
 def _fix_origin_visits(visits):
     good_visits = []
     for visit in visits:
         visit = visit.copy()
         if 'type' not in visit:
             if isinstance(visit['origin'], dict) and 'type' in visit['origin']:
                 # Very old version of the schema: visits did not have a type,
                 # but their 'origin' field was a dict with a 'type' key.
                 visit['type'] = visit['origin']['type']
             else:
                 # Very very old version of the schema: 'type' is missing,
                 # so there is nothing we can do to fix it.
                 raise ValueError('Got an origin_visit too old to be replayed.')
         if isinstance(visit['origin'], dict):
             # Old version of the schema: visit['origin'] was a dict.
             visit['origin'] = visit['origin']['url']
         good_visits.append(visit)
     return good_visits
 
 
 def fix_objects(object_type, objects):
     """Converts a possibly old object from the journal to its current
     expected format.
 
     List of conversions:
 
     Empty author name/email in PyPI releases:
 
     >>> from pprint import pprint
     >>> date = {
     ...     'timestamp': {
     ...         'seconds': 1565096932,
     ...         'microseconds': 0,
     ...     },
     ...     'offset': 0,
     ... }
     >>> pprint(fix_objects('revision', [{
     ...     'author': {'email': '', 'fullname': b'', 'name': ''},
     ...     'committer': {'email': '', 'fullname': b'', 'name': ''},
     ...     'date': date,
     ...     'committer_date': date,
     ... }]))
     [{'author': {'email': b'', 'fullname': b'', 'name': b''},
       'committer': {'email': b'', 'fullname': b'', 'name': b''},
       'committer_date': {'offset': 0,
                          'timestamp': {'microseconds': 0, 'seconds': 1565096932}},
       'date': {'offset': 0,
                'timestamp': {'microseconds': 0, 'seconds': 1565096932}}}]
 
     Fix type of 'transplant_source' extra headers:
 
     >>> revs = fix_objects('revision', [{
     ...     'author': {'email': '', 'fullname': b'', 'name': ''},
     ...     'committer': {'email': '', 'fullname': b'', 'name': ''},
     ...     'date': date,
     ...     'committer_date': date,
     ...     'metadata': {
     ...         'extra_headers': [
     ...             ['time_offset_seconds', b'-3600'],
     ...             ['transplant_source', '29c154a012a70f49df983625090434587622b39e']
     ...     ]}
     ... }])
     >>> pprint(revs[0]['metadata']['extra_headers'])
     [['time_offset_seconds', b'-3600'],
      ['transplant_source', b'29c154a012a70f49df983625090434587622b39e']]
 
     Filter out revisions with invalid dates:
 
     >>> from copy import deepcopy
     >>> invalid_date1 = deepcopy(date)
     >>> invalid_date1['timestamp']['microseconds'] = 1000000000  # > 10^6
     >>> fix_objects('revision', [{
     ...     'author': {'email': '', 'fullname': b'', 'name': b''},
     ...     'committer': {'email': '', 'fullname': b'', 'name': b''},
     ...     'date': invalid_date1,
     ...     'committer_date': date,
     ... }])
     []
 
     >>> invalid_date2 = deepcopy(date)
     >>> invalid_date2['timestamp']['seconds'] = 2**70  # > 10^63
     >>> fix_objects('revision', [{
     ...     'author': {'email': '', 'fullname': b'', 'name': b''},
     ...     'committer': {'email': '', 'fullname': b'', 'name': b''},
     ...     'date': invalid_date2,
     ...     'committer_date': date,
     ... }])
     []
 
     >>> invalid_date3 = deepcopy(date)
     >>> invalid_date3['offset'] = 2**20  # > 10^15
     >>> fix_objects('revision', [{
     ...     'author': {'email': '', 'fullname': b'', 'name': b''},
     ...     'committer': {'email': '', 'fullname': b'', 'name': b''},
     ...     'date': date,
     ...     'committer_date': invalid_date3,
     ... }])
     []
 
 
     `visit['origin']` is a dict instead of an URL:
 
     >>> pprint(fix_objects('origin_visit', [{
     ...     'origin': {'url': 'http://foo'},
     ...     'type': 'git',
     ... }]))
     [{'origin': 'http://foo', 'type': 'git'}]
 
     `visit['type']` is missing , but `origin['visit']['type']` exists:
 
     >>> pprint(fix_objects('origin_visit', [
     ...     {'origin': {'type': 'hg', 'url': 'http://foo'}
     ... }]))
     [{'origin': 'http://foo', 'type': 'hg'}]
     """  # noqa
 
     if object_type == 'revision':
         objects = _fix_revisions(objects)
     elif object_type == 'origin_visit':
         objects = _fix_origin_visits(objects)
     return objects
 
 
 def _insert_objects(object_type, objects, storage):
     objects = fix_objects(object_type, objects)
     if object_type == 'content':
         # TODO: insert 'content' in batches
         for object_ in objects:
             try:
                 storage.content_add_metadata([object_])
             except HashCollision as e:
                 logger.error('Hash collision: %s', e.args)
     elif object_type in ('directory', 'revision', 'release',
                          'snapshot', 'origin'):
         # TODO: split batches that are too large for the storage
         # to handle?
         method = getattr(storage, object_type + '_add')
         method(objects)
     elif object_type == 'origin_visit':
         for visit in objects:
             storage.origin_add_one({'url': visit['origin']})
             if 'metadata' not in visit:
                 visit['metadata'] = None
         storage.origin_visit_upsert(objects)
     else:
         logger.warning('Received a series of %s, this should not happen',
                        object_type)
 
 
 def is_hash_in_bytearray(hash_, array, nb_hashes, hash_size=SHA1_SIZE):
     """
     Checks if the given hash is in the provided `array`. The array must be
     a *sorted* list of sha1 hashes, and contain `nb_hashes` hashes
     (so its size must by `nb_hashes*hash_size` bytes).
 
     Args:
         hash_ (bytes): the hash to look for
         array (bytes): a sorted concatenated array of hashes (may be of
             any type supporting slice indexing, eg. :py:cls:`mmap.mmap`)
         nb_hashes (int): number of hashes in the array
         hash_size (int): size of a hash (defaults to 20, for SHA1)
 
     Example:
 
     >>> import os
     >>> hash1 = os.urandom(20)
     >>> hash2 = os.urandom(20)
     >>> hash3 = os.urandom(20)
     >>> array = b''.join(sorted([hash1, hash2]))
     >>> is_hash_in_bytearray(hash1, array, 2)
     True
     >>> is_hash_in_bytearray(hash2, array, 2)
     True
     >>> is_hash_in_bytearray(hash3, array, 2)
     False
     """
     if len(hash_) != hash_size:
         raise ValueError('hash_ does not match the provided hash_size.')
 
     def get_hash(position):
         return array[position*hash_size:(position+1)*hash_size]
 
     # Regular dichotomy:
     left = 0
     right = nb_hashes
     while left < right-1:
         middle = int((right+left)/2)
         pivot = get_hash(middle)
         if pivot == hash_:
             return True
         elif pivot < hash_:
             left = middle
         else:
             right = middle
     return get_hash(left) == hash_
 
 
 @contextmanager
 def retry(max_retries):
     lasterror = None
     for i in range(max_retries):
         try:
             yield
             break
         except Exception as exc:
             lasterror = exc
     else:
         raise lasterror
 
 
 def copy_object(obj_id, src, dst, max_retries=3):
     try:
         with statsd.timed(CONTENT_DURATION_METRIC, tags={'request': 'get'}):
             with retry(max_retries):
                 obj = src.get(obj_id)
                 logger.debug('retrieved %s', hash_to_hex(obj_id))
 
         with statsd.timed(CONTENT_DURATION_METRIC, tags={'request': 'put'}):
             with retry(max_retries):
                 dst.add(obj, obj_id=obj_id, check_presence=False)
                 logger.debug('copied %s', hash_to_hex(obj_id))
         statsd.increment(CONTENT_OPERATIONS_METRIC)
         statsd.increment(CONTENT_BYTES_METRIC, len(obj))
     except Exception:
         obj = ''
         logger.error('Failed to copy %s', hash_to_hex(obj_id))
         raise
     return len(obj)
 
 
 def process_replay_objects_content(all_objects, *, src, dst,
                                    exclude_fn=None):
     """
     Takes a list of records from Kafka (see
     :py:func:`swh.journal.client.JournalClient.process`) and copies them
     from the `src` objstorage to the `dst` objstorage, if:
 
     * `obj['status']` is `'visible'`
     * `exclude_fn(obj)` is `False` (if `exclude_fn` is provided)
 
     Args:
         all_objects Dict[str, List[dict]]: Objects passed by the Kafka client.
             Most importantly, `all_objects['content'][*]['sha1']` is the
             sha1 hash of each content
         src: An object storage (see :py:func:`swh.objstorage.get_objstorage`)
         dst: An object storage (see :py:func:`swh.objstorage.get_objstorage`)
         exclude_fn Optional[Callable[dict, bool]]: Determines whether
             an object should be copied.
 
     Example:
 
     >>> from swh.objstorage import get_objstorage
     >>> src = get_objstorage('memory', {})
     >>> dst = get_objstorage('memory', {})
     >>> id1 = src.add(b'foo bar')
     >>> id2 = src.add(b'baz qux')
     >>> kafka_partitions = {
     ...     'content': [
     ...         {
     ...             'sha1': id1,
     ...             'status': 'visible',
     ...         },
     ...         {
     ...             'sha1': id2,
     ...             'status': 'visible',
     ...         },
     ...     ]
     ... }
     >>> process_replay_objects_content(
     ...     kafka_partitions, src=src, dst=dst,
     ...     exclude_fn=lambda obj: obj['sha1'] == id1)
     >>> id1 in dst
     False
     >>> id2 in dst
     True
     """
     vol = []
     nb_skipped = 0
     t0 = time()
 
     for (object_type, objects) in all_objects.items():
         if object_type != 'content':
             logger.warning(
                 'Received a series of %s, this should not happen',
                 object_type)
             continue
         for obj in objects:
             obj_id = obj[ID_HASH_ALGO]
             if obj['status'] != 'visible':
                 nb_skipped += 1
                 logger.debug('skipped %s (status=%s)',
                              hash_to_hex(obj_id), obj['status'])
             elif exclude_fn and exclude_fn(obj):
                 nb_skipped += 1
                 logger.debug('skipped %s (manually excluded)',
                              hash_to_hex(obj_id))
             else:
                 vol.append(copy_object(obj_id, src, dst))
 
     dt = time() - t0
     logger.info(
         'processed %s content objects in %.1fsec '
         '(%.1f obj/sec, %.1fMB/sec) - %d failures - %d skipped',
         len(vol), dt,
         len(vol)/dt,
         sum(vol)/1024/1024/dt,
         len([x for x in vol if not x]),
         nb_skipped)
+
+    if notify:
+        notify('WATCHDOG=1')
diff --git a/swh/journal/tests/test_kafka_writer.py b/swh/journal/tests/test_kafka_writer.py
index 07a8b3a..bab6639 100644
--- a/swh/journal/tests/test_kafka_writer.py
+++ b/swh/journal/tests/test_kafka_writer.py
@@ -1,138 +1,144 @@
 # Copyright (C) 2018-2019 The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 from collections import defaultdict
 import datetime
 
 from confluent_kafka import Consumer, KafkaException
 from subprocess import Popen
 from typing import Tuple
 
 from swh.storage import get_storage
 
 from swh.journal.writer.kafka import KafkaJournalWriter
 from swh.journal.serializers import (
     kafka_to_key, kafka_to_value
 )
 
 from .conftest import OBJECT_TYPE_KEYS
 
 
 def assert_written(consumer, kafka_prefix, expected_messages):
     consumed_objects = defaultdict(list)
 
     fetched_messages = 0
     retries_left = 1000
 
     while fetched_messages < expected_messages:
         if retries_left == 0:
             raise ValueError('Timed out fetching messages from kafka')
 
         msg = consumer.poll(timeout=0.01)
 
         if not msg:
             retries_left -= 1
             continue
 
         error = msg.error()
         if error is not None:
             if error.fatal():
                 raise KafkaException(error)
             retries_left -= 1
             continue
 
         fetched_messages += 1
         consumed_objects[msg.topic()].append(
             (kafka_to_key(msg.key()), kafka_to_value(msg.value()))
         )
 
     for (object_type, (key_name, objects)) in OBJECT_TYPE_KEYS.items():
         topic = kafka_prefix + '.' + object_type
         (keys, values) = zip(*consumed_objects[topic])
         if key_name:
             assert list(keys) == [object_[key_name] for object_ in objects]
         else:
             pass  # TODO
 
         if object_type == 'origin_visit':
             for value in values:
                 del value['visit']
         elif object_type == 'content':
             for value in values:
                 del value['ctime']
 
         for object_ in objects:
             assert object_ in values
 
 
 def test_kafka_writer(
         kafka_prefix: str,
         kafka_server: Tuple[Popen, int],
         consumer: Consumer):
     kafka_prefix += '.swh.journal.objects'
 
     config = {
         'brokers': ['localhost:%d' % kafka_server[1]],
         'client_id': 'kafka_writer',
         'prefix': kafka_prefix,
+        'producer_config': {
+            'message.max.bytes': 100000000,
+        }
     }
 
     writer = KafkaJournalWriter(**config)
 
     expected_messages = 0
 
     for (object_type, (_, objects)) in OBJECT_TYPE_KEYS.items():
         for (num, object_) in enumerate(objects):
             if object_type == 'origin_visit':
                 object_ = {**object_, 'visit': num}
             if object_type == 'content':
                 object_ = {**object_, 'ctime': datetime.datetime.now()}
             writer.write_addition(object_type, object_)
             expected_messages += 1
 
     assert_written(consumer, kafka_prefix, expected_messages)
 
 
 def test_storage_direct_writer(
         kafka_prefix: str,
         kafka_server: Tuple[Popen, int],
         consumer: Consumer):
     kafka_prefix += '.swh.journal.objects'
 
     config = {
         'brokers': ['localhost:%d' % kafka_server[1]],
         'client_id': 'kafka_writer',
         'prefix': kafka_prefix,
+        'producer_config': {
+            'message.max.bytes': 100000000,
+        }
     }
 
     storage = get_storage('memory', journal_writer={
-        'cls': 'kafka', 'args': config,
+        'cls': 'kafka', **config,
     })
 
     expected_messages = 0
 
     for (object_type, (_, objects)) in OBJECT_TYPE_KEYS.items():
         method = getattr(storage, object_type + '_add')
         if object_type in ('content', 'directory', 'revision', 'release',
                            'snapshot', 'origin'):
             if object_type == 'content':
                 objects = [{**obj, 'data': b''} for obj in objects]
             method(objects)
             expected_messages += len(objects)
         elif object_type in ('origin_visit',):
             for object_ in objects:
                 object_ = object_.copy()
                 origin_url = object_.pop('origin')
                 storage.origin_add_one({'url': origin_url})
                 visit = method(origin=origin_url, date=object_.pop('date'),
                                type=object_.pop('type'))
                 expected_messages += 1
                 visit_id = visit['visit']
                 storage.origin_visit_update(origin_url, visit_id, **object_)
                 expected_messages += 1
         else:
             assert False, object_type
 
     assert_written(consumer, kafka_prefix, expected_messages)
diff --git a/swh/journal/writer/__init__.py b/swh/journal/writer/__init__.py
index 7540c40..bd12467 100644
--- a/swh/journal/writer/__init__.py
+++ b/swh/journal/writer/__init__.py
@@ -1,20 +1,28 @@
 # Copyright (C) 2019 The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
+import warnings
+
+
+def get_journal_writer(cls, **kwargs):
+    if 'args' in kwargs:
+        warnings.warn(
+            'Explicit "args" key is deprecated, use keys directly instead.',
+            DeprecationWarning)
+        kwargs = kwargs['args']
 
-def get_journal_writer(cls, args={}):
     if cls == 'inmemory':  # FIXME: Remove inmemory in due time
         import warnings
         warnings.warn("cls = 'inmemory' is deprecated, use 'memory' instead",
                       DeprecationWarning)
         cls = 'memory'
     if cls == 'memory':
         from .inmemory import InMemoryJournalWriter as JournalWriter
     elif cls == 'kafka':
         from .kafka import KafkaJournalWriter as JournalWriter
     else:
         raise ValueError('Unknown journal writer class `%s`' % cls)
 
-    return JournalWriter(**args)
+    return JournalWriter(**kwargs)
diff --git a/swh/journal/writer/kafka.py b/swh/journal/writer/kafka.py
index 70ed938..d4728fb 100644
--- a/swh/journal/writer/kafka.py
+++ b/swh/journal/writer/kafka.py
@@ -1,110 +1,114 @@
 # Copyright (C) 2019 The Software Heritage developers
 # See the AUTHORS file at the top-level directory of this distribution
 # License: GNU General Public License version 3, or any later version
 # See top-level LICENSE file for more information
 
 import logging
 
 from confluent_kafka import Producer, KafkaException
 
 from swh.model.hashutil import DEFAULT_ALGORITHMS
 from swh.model.model import BaseModel
 
 from swh.journal.serializers import key_to_kafka, value_to_kafka
 
 logger = logging.getLogger(__name__)
 
 
 class KafkaJournalWriter:
     """This class is instantiated and used by swh-storage to write incoming
     new objects to Kafka before adding them to the storage backend
     (eg. postgresql) itself."""
-    def __init__(self, brokers, prefix, client_id):
+    def __init__(self, brokers, prefix, client_id, producer_config=None):
         self._prefix = prefix
 
         if isinstance(brokers, str):
             brokers = [brokers]
 
+        if not producer_config:
+            producer_config = {}
+
         self.producer = Producer({
             'bootstrap.servers': ','.join(brokers),
             'client.id': client_id,
             'on_delivery': self._on_delivery,
             'error_cb': self._error_cb,
             'logger': logger,
             'enable.idempotence': 'true',
+            **producer_config,
         })
 
     def _error_cb(self, error):
         if error.fatal():
             raise KafkaException(error)
         logger.info('Received non-fatal kafka error: %s', error)
 
     def _on_delivery(self, error, message):
         if error is not None:
             self._error_cb(error)
 
     def send(self, topic, key, value):
         self.producer.produce(
             topic=topic,
             key=key_to_kafka(key),
             value=value_to_kafka(value),
         )
 
         # Need to service the callbacks regularly by calling poll
         self.producer.poll(0)
 
     def flush(self):
         self.producer.flush()
 
     def _get_key(self, object_type, object_):
         if object_type in ('revision', 'release', 'directory', 'snapshot'):
             return object_['id']
         elif object_type == 'content':
             return object_['sha1']  # TODO: use a dict of hashes
         elif object_type == 'skipped_content':
             return {
                 hash: object_[hash]
                 for hash in DEFAULT_ALGORITHMS
             }
         elif object_type == 'origin':
             return {'url': object_['url']}
         elif object_type == 'origin_visit':
             return {
                 'origin': object_['origin'],
                 'date': str(object_['date']),
             }
         else:
             raise ValueError('Unknown object type: %s.' % object_type)
 
     def _sanitize_object(self, object_type, object_):
         if object_type == 'origin_visit':
             return {
                 **object_,
                 'date': str(object_['date']),
             }
         elif object_type == 'origin':
             assert 'id' not in object_
         return object_
 
     def write_addition(self, object_type, object_, flush=True):
         """Write a single object to the journal"""
         if isinstance(object_, BaseModel):
             object_ = object_.to_dict()
         topic = '%s.%s' % (self._prefix, object_type)
         key = self._get_key(object_type, object_)
         object_ = self._sanitize_object(object_type, object_)
         logger.debug('topic: %s, key: %s, value: %s', topic, key, object_)
         self.send(topic, key=key, value=object_)
 
         if flush:
             self.flush()
 
     write_update = write_addition
 
     def write_additions(self, object_type, objects, flush=True):
         """Write a set of objects to the journal"""
         for object_ in objects:
             self.write_addition(object_type, object_, flush=False)
 
         if flush:
             self.flush()
diff --git a/version.txt b/version.txt
index 23b5ad4..c695c4f 100644
--- a/version.txt
+++ b/version.txt
@@ -1 +1 @@
-v0.0.21-0-g3bb1d77
\ No newline at end of file
+v0.0.23-0-ga9fb7c7
\ No newline at end of file