diff --git a/docs-main/docs.json b/docs-main/docs.json
index e882ca558..5c4680ac9 100644
--- a/docs-main/docs.json
+++ b/docs-main/docs.json
@@ -850,7 +850,8 @@
"sdks-tools/development-tools/pqs",
"sdks-tools/development-tools/pqs/configure",
"sdks-tools/development-tools/pqs/operate",
- "sdks-tools/development-tools/pqs/optimize"
+ "sdks-tools/development-tools/pqs/optimize",
+ "sdks-tools/development-tools/pqs/troubleshoot"
]
}
]
diff --git a/docs-main/images/docs_website/20250408-dashboard-1-contracts-active.png b/docs-main/images/docs_website/20250408-dashboard-1-contracts-active.png
new file mode 100644
index 000000000..895551710
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-1-contracts-active.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-1-contracts-churn.png b/docs-main/images/docs_website/20250408-dashboard-1-contracts-churn.png
new file mode 100644
index 000000000..9c5345a7e
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-1-contracts-churn.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-events-breakdown.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-events-breakdown.png
new file mode 100644
index 000000000..77519f717
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-events-breakdown.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-ingested-counts.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-ingested-counts.png
new file mode 100644
index 000000000..0fdbff0ec
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-ingested-counts.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-throughputs.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-throughputs.png
new file mode 100644
index 000000000..224470d90
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-throughputs.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-tx-and-events.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-tx-and-events.png
new file mode 100644
index 000000000..a5ef24112
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-tx-and-events.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-tx-lag.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-tx-lag.png
new file mode 100644
index 000000000..b1c0b8b8c
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-tx-lag.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-waitpoints-streaming.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-waitpoints-streaming.png
new file mode 100644
index 000000000..6cf2bfc2b
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-waitpoints-streaming.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-2-throughput-watermark-history.png b/docs-main/images/docs_website/20250408-dashboard-2-throughput-watermark-history.png
new file mode 100644
index 000000000..c70b09174
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-2-throughput-watermark-history.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-3-queue-sizes-events-1.png b/docs-main/images/docs_website/20250408-dashboard-3-queue-sizes-events-1.png
new file mode 100644
index 000000000..4bc5c93e0
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-3-queue-sizes-events-1.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-3-queue-sizes-events.png b/docs-main/images/docs_website/20250408-dashboard-3-queue-sizes-events.png
new file mode 100644
index 000000000..ea6462863
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-3-queue-sizes-events.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-4-latency-jdbc-connection.png b/docs-main/images/docs_website/20250408-dashboard-4-latency-jdbc-connection.png
new file mode 100644
index 000000000..184f075c5
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-4-latency-jdbc-connection.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-4-latency-total-tx-handling.png b/docs-main/images/docs_website/20250408-dashboard-4-latency-total-tx-handling.png
new file mode 100644
index 000000000..34a40bbb5
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-4-latency-total-tx-handling.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-5-jvm-metrics.png b/docs-main/images/docs_website/20250408-dashboard-5-jvm-metrics.png
new file mode 100644
index 000000000..1532f5f50
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-5-jvm-metrics.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-6-postgres-metrics-1.png b/docs-main/images/docs_website/20250408-dashboard-6-postgres-metrics-1.png
new file mode 100644
index 000000000..8e4b92f0c
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-6-postgres-metrics-1.png differ
diff --git a/docs-main/images/docs_website/20250408-dashboard-6-postgres-metrics-2.png b/docs-main/images/docs_website/20250408-dashboard-6-postgres-metrics-2.png
new file mode 100644
index 000000000..080930301
Binary files /dev/null and b/docs-main/images/docs_website/20250408-dashboard-6-postgres-metrics-2.png differ
diff --git a/docs-main/images/docs_website/grafana-pqs-dashboard.png b/docs-main/images/docs_website/grafana-pqs-dashboard.png
new file mode 100644
index 000000000..2c0b752c1
Binary files /dev/null and b/docs-main/images/docs_website/grafana-pqs-dashboard.png differ
diff --git a/docs-main/images/docs_website/trace-advance-watermark.png b/docs-main/images/docs_website/trace-advance-watermark.png
new file mode 100644
index 000000000..6a38f1a3d
Binary files /dev/null and b/docs-main/images/docs_website/trace-advance-watermark.png differ
diff --git a/docs-main/images/docs_website/trace-consumer.png b/docs-main/images/docs_website/trace-consumer.png
new file mode 100644
index 000000000..deca18f5e
Binary files /dev/null and b/docs-main/images/docs_website/trace-consumer.png differ
diff --git a/docs-main/images/docs_website/trace-export-batch.png b/docs-main/images/docs_website/trace-export-batch.png
new file mode 100644
index 000000000..14f3e48c7
Binary files /dev/null and b/docs-main/images/docs_website/trace-export-batch.png differ
diff --git a/docs-main/sdks-tools/development-tools/pqs/configure.mdx b/docs-main/sdks-tools/development-tools/pqs/configure.mdx
index 5b4167c93..7b88839fb 100644
--- a/docs-main/sdks-tools/development-tools/pqs/configure.mdx
+++ b/docs-main/sdks-tools/development-tools/pqs/configure.mdx
@@ -1,34 +1,45 @@
---
title: "Configure"
-description: "Component how-to: Configure"
+description: "Configure PQS data sources, filters, and encoding options."
---
-{/* COPIED_START source="docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/configure.rst" hash="d1cc365e" */}
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/configure.rst" hash="e7832735" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
PQS ascertains its configuration from the following sources, in order of priority:
- HOCON configuration files (`--config` argument)
+
- command-line arguments
+
- Java system properties (`-D` arguments to `java` command)
+
- environment variables
-Consult the command `./scribe.jar pipeline --help-verbose` for further information on individual configuration items, and the conventions used to specify them in the above forms.
+Consult the command `./scribe.jar pipeline --help-verbose` for further information on individual configuration
+items, and the conventions used to specify them in the above forms.
-PQS sets its configuration at startup. It does not perform dynamic configuration updates, so making a configuration change (such as adding a new party, new template, or new interface to filters) requires a restart. Deploying new Daml packages (DARs) to the Participant Node does **not** require restarting PQS. When PQS encounters an unknown package while processing events, it fetches the missing packages from the Participant Node automatically. This may pause ingestion for up to tens of seconds.
+PQS sets its configuration at startup. It does not perform dynamic configuration updates, so making a configuration
+change (such as adding a new party, or new interface to filters) requires a restart.
-
+
+Deploying new Daml packages (DARs) to the Participant Node does not require restarting PQS. When PQS encounters an unknown package while processing events, it fetches the missing packages from the Participant Node. This could pause ingestion for up to tens of seconds. See [Dynamic Daml package reload](/sdks-tools/development-tools/pqs/operate#dynamic-daml-package-reload) for details.
+
-PQS will not go back in time and recover history, but only move forward by consuming new transactions it has not previously seen. It is important that scope only be expanded when it is known to have no prior history, at the point in time that PQS was stopped. Otherwise, a re-seed operation will be required to reinitialize from an empty datastore.
+PQS will not go back in time and recover history, but only move forward by consuming new transactions it has not
+previously seen. It is important that scope only be expanded when it is known to have no prior history, at the point
+in time that PQS was stopped. Otherwise, a re-seed operation will be required to reinitialize from an empty
+datastore.
-
-
-The following command connects to a non-auth ledger and replicates the latest state of the ledger (excluding prior-history) from the perspective of the supplied Daml party. It uses the ledger source and supplied database:
+The following command connects to a non-auth ledger and replicates the latest state of the ledger (excluding
+prior-history) from the perspective of the supplied Daml party. It uses the ledger source and supplied database:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--pipeline-filter-parties "Alice::* | Bob::*" \
--pipeline-ledger-start Latest \
@@ -39,85 +50,120 @@ $ ./scribe.jar pipeline ledger postgres-document \
--target-postgres-database postgres
```
-Consult `pqs-references-configuration-options` how to configure each command.
+Consult *Configuration options* how to configure each command.
## Transactions data source
-To understand how PQS stores data, you need to understand the `da-ledgers`. In simple terms, the Daml ledger is composed of a sequence of transactions, which contain events. Events can be:
+To understand how PQS stores data, you need to understand the [Ledger Model](/overview/learn/ledger-model). In simple terms,
+the Daml ledger is composed of a sequence of transactions, which contain events. Events can be:
- Create: creation of contracts / interface views / divulgences
+
- Exercise: of a choice of contracts / interface views
-- Archive: end of the lifetime of contracts / interface views
-
+- Archive: end of the lifetime of contracts / interface views
-When defining the scope of ledger data being stored, it is important to understand the implications of the data source and the filters applied. The data source and filters determine the data that is available to the SQL functions (see `pqs-references-sql-api`), and this **cannot be changed**. Since a change in scope will result in a change to the breadth of data being stored, a re-seed is required to widen or narrow the scope of the data. The only exception to this is where you widen the scope into an area, for example new templates, new parties, that you know has no historical data, in which case a re-seed is not required. Or, operators may also use the reset function to roll-back the datastore to a prior state where this was true.
+When defining the scope of ledger data being stored, it is important to understand the implications of the data
+source and the filters applied. The data source and filters determine the data that is available to the
+SQL functions (see [SQL API](/appdev/reference/pqs-sql-reference)), and this **cannot be changed**. Since a change in scope will result in a
+change to the breadth of data being stored, a re-seed is required to widen or narrow the scope of the data. The
+only exception to this is where you widen the scope into an area, for example new templates, new parties, that you
+know has no historical data, in which case a re-seed is not required. Or, operators may also use the reset function
+to roll-back the datastore to a prior state where this was true.
-
-
-PQS can run in two modes as specified by the `--pipeline-datasource` configuration. The following table shows the differences between the two modes, in terms of data availability via the respective SQL functions:
-
-| Data / Mode | TransactionStream | TransactionTreeStream |
-|---------------------------------------------|-------------------|-----------------------|
-| `creates()` of contracts | ✓ | ✓ |
-| `exercises()` on contracts | ✗ | ✓ |
-| `archives()` of contracts | ✓ | ✓ |
-| `creates()` of interfaces | ✓ | ✓ |
-| `exercises()` on interfaces | ✗ | ✓ |
-| `archives()` of interfaces | ✓ | ✓ |
-| `creates()` of divulgences | ✗ | ✓ |
-| Transient (create-archive in a transaction) | ✗ | ✓ |
-| Stakeholders & witnesses | ✓ | ✓ |
-| Choice controllers | ✗ | ✓ |
-| **Note** | | |
-| Default | ✓ | ✗ |
-| Data size | Smaller | Larger |
+PQS can run in two modes as specified by the `--pipeline-datasource` configuration. The following table shows the
+differences between the two modes, in terms of data availability via the respective SQL functions:
+
+| Data / Mode | TransactionStream | TransactionTreeStream |
+| --- | --- | --- |
+| `creates()` of contracts | ✓ | ✓ |
+| `exercises()` on contracts | ✗ | ✓ |
+| `archives()` of contracts | ✓ | ✓ |
+| `creates()` of interfaces | ✓ | ✓ |
+| `exercises()` on interfaces | ✗ | ✓ |
+| `archives()` of interfaces | ✓ | ✓ |
+| `creates()` of divulgences | ✗ | ✓ |
+| Transient (create-archive in a transaction) | ✗ | ✓ |
+| Stakeholders & witnesses | ✓ | ✓ |
+| Choice controllers | ✗ | ✓ |
+| **Note** | | |
+| Default | ✓ | ✗ |
+| Data size | Smaller | Larger |
## Contract filtering
-`--pipeline-filter-contracts` specifies a filter expression to determine which the Daml templates, interface views and choices to include. A filter expression is a simple wildcard inclusion (`*`) with basic boolean logic (`&` `!` `|` `(` `)`), where whitespace is ignored. For example:
+`--pipeline-filter-contracts` specifies a filter expression to determine which the Daml templates, interface views
+and choices to include. A filter expression is a simple wildcard inclusion (`*`) with basic boolean logic (`&`
+`!` `|` `(` `)`), where whitespace is ignored. For example:
- `*`: everything (default)
+
- `pkg:*`: everything in this package
+
- `pkg@=1`: everything in this package that has major version `1.x.x`
+
- `pkg@>=1.2.3`: everything in this package starting with version `1.2.3` inclusively
+
- `foo@1.2.3|bar@3.2.1`: everything in pinpointed packages `foo` at version `1.2.3` and package `bar` at version `3.2.1`
+
- `pkg:a.b.c.Bar`: just this one fully qualified name for template `Bar`
+
- `a.b.c.*`: all members of the `a.b.c` namespace
+
- `* & !pkg:a.b.c.Bar`: everything except this one fully qualified name
+
- `(a.b.c.Foo | a.b.c.Bar)`: these two fully qualified names
+
- `(a.b.c.* & !(a.b.c.Foo | a.b.c.Bar) | g.e.f.Baz)`: everything in `a.b.c` except for `Foo` and `Bar`, and also include `g.e.f.Baz`
+
- `a.b.c.Foo & a.b.c.Bar`: error (the identifier can't be both)
-There are further conditions placed upon the filtering of templates and interfaces to avoid potential ambiguity. It is required to include any filter for:
+There are further conditions placed upon the filtering of templates and interfaces to avoid potential ambiguity.
+It is required to include any filter for:
- All interface views of included templates
+
- All templates of included interface views
## Party filtering
-Similarly, the `--pipeline-filter-parties` option specifies a filter expression to determine which parties to supply data for. For example:
+Similarly, the `--pipeline-filter-parties` option specifies a filter expression to determine which
+parties to supply data for. For example:
- `*`: everything (default)
+
- `Alice::* | Bob::*`: any party with an `Alice` or `Bob` hint
+
- `Alice::122055fc4b190e3ff438587b699495a4b6388e911e2305f7e013af160f49a76080ab`: just this one party
+
- `* & !Alice::*`: all parties except those with an `Alice` hint
+
- `Alice* | Bob* | (Charlie* & !(Charlie3::*))`: `Alice` and `Bob` parties, as well as `Charlie` except `Charlie3`
-When the ledger requires authentication, this filter applies within the scope of parties for which PQS' Ledger API user has access. Naturally, the `--pipeline-filter-parties` cannot be used to access data for parties for which the user is not authorized.
+When the ledger requires authentication, this filter applies within the scope of parties for which PQS' Ledger API
+user has access. Naturally, the `--pipeline-filter-parties` cannot be used to access data for parties for which the
+user is not authorized.
## Explicit contract disclosure
-`--pipeline-filter-metadata` specifies an inclusion filter expression to determine the Daml templates and interface views to capture metadata for. Same syntax as section above applies. Captured data will be available in the `metadata` column (`bytea` PostgreSQL type) in the query functions output. This column stores the contents of `created_event_blob` (see `stakeholder-contract-share`) from the respective event.
+`--pipeline-filter-metadata` specifies an inclusion filter expression to determine the Daml templates and interface
+views to capture metadata for. Same syntax as section above applies. Captured data will be available in the
+`metadata` column (`bytea` PostgreSQL type) in the query functions output. This column stores the contents of
+`created_event_blob` (see [How do stakeholders disclose contracts to submitters?](/appdev/deep-dives/explicit-contract-disclosure#how-do-stakeholders-disclose-contracts-to-submitters)) from the respective event.
## JSON encoding configuration
-By default PQS stores the payloads in JSON format using strings for numeric types. This is the default because numbers in JSON are backed by double-precision floating point numbers, which can lead to loss of precision for large integers. If you want to store numbers as integers and can tolerate loss of precision in numbers, you can use the `--target-encoding-numericasstring` and `--target-encoding-int64asstring` options to change the default behavior.
+By default PQS stores the payloads in JSON format using strings for numeric types. This is the default because
+numbers in JSON are backed by double-precision floating point numbers, which can lead to loss of precision for
+large integers. If you want to store numbers as integers and can tolerate loss of precision in numbers, you can use the
+`--target-encoding-numericasstring` and `--target-encoding-int64asstring` options to change the default behavior.
-Another option allows to store nullable fields as JSON nulls instead of omitting nullable fields from the records. PQS stores nullable fields as JSON nulls by default, but you can change this behavior using the `--target-encoding-excludenulls` option.
+Another option allows to store nullable fields as JSON nulls instead of omitting nullable fields from the records.
+PQS stores nullable fields as JSON nulls by default, but you can change this behavior using the
+`--target-encoding-excludenulls` option.
{/* COPIED_END */}
diff --git a/docs-main/sdks-tools/development-tools/pqs/operate.mdx b/docs-main/sdks-tools/development-tools/pqs/operate.mdx
index cd23a229a..29bb5199b 100644
--- a/docs-main/sdks-tools/development-tools/pqs/operate.mdx
+++ b/docs-main/sdks-tools/development-tools/pqs/operate.mdx
@@ -3,37 +3,45 @@ title: "Operate"
description: "Run, observe, recover, and secure a PQS deployment."
---
+
## Run and prune
-{/* COPIED_START source="docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/operate.rst" hash="f00146d1" */}
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/operate.rst" hash="ee51346b" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
To run PQS you need the following:
- PostgreSQL database server
+
- Daml Sandbox or Canton Participant Node as the source of ledger data
+
- Any access tokens or TLS certificates required by the above
+
- PQS' `scribe.jar` or Docker image
### Running PQS
PQS application mostly runs as a long-running process, but also includes several user-interactive commands:
-| Command | Description |
-|--------------------------------------------|-----------------------------------------------------------------|
-| `pipeline ledger postgres-document` | Initiate continuous ledger data export |
-| `datastore postgres-document schema show` | Infer required database schema, display it and quit |
+| Command | Description |
+| --- | --- |
+| `pipeline ledger postgres-document` | Initiate continuous ledger data export |
+| `datastore postgres-document schema show` | Infer required database schema, display it and quit |
| `datastore postgres-document schema apply` | Infer required database schema, apply it to data store and quit |
-| `datastore postgres-document prune` | Prune transactions to a given offset inclusively and quit |
+| `datastore postgres-document prune` | Prune transactions to a given offset inclusively and quit |
-Consult `pqs-references-configuration-options` how to configure each command.
+Consult *Configuration options* how to configure each command.
-PQS pipeline is crash friendly and restarts automatically. See `pqs-ledger-streaming-and-recovery` for more details on how PQS recovers from a crash.
+PQS pipeline is crash friendly and restarts automatically. See [Ledger streaming & recovery](#ledger-streaming-recovery) for more
+details on how PQS recovers from a crash.
### Getting help
-Exploring commands and parameters is easiest via the `--help` (and `--help-verbose`) arguments: For example, if you are running a downloaded `.jar` file:
+Exploring commands and parameters is easiest via the `--help` (and `--help-verbose`) arguments: For example, if
+you are running a downloaded `.jar` file:
-``` text
+```text
$ ./scribe.jar --help
Usage: scribe COMMAND
@@ -53,7 +61,7 @@ Run 'scribe COMMAND --help[-verbose]' for more information on a command.
Or similarly, using Docker:
-``` text
+```text
$ docker run -it europe-docker.pkg.dev/da-images/public/docker/participant-query-store:0.6.14 --help
Picked up JAVA_TOOL_OPTIONS: -javaagent:/open-telemetry.jar
Usage: scribe COMMAND
@@ -69,165 +77,195 @@ Run 'scribe COMMAND --help[-verbose]' for more information on a command.
### History slicing
-As described in `pqs-ledger-streaming-and-recovery` you can use PQS with `--pipeline-ledger-start` and `--pipeline-ledger-stop` to ask for the slice of the history you want. There are some constraints on start and stop offsets which cause PQS to fail-fast if they are violated.
+As described in [Ledger streaming & recovery](#ledger-streaming-recovery) you can use PQS with `--pipeline-ledger-start` and
+`--pipeline-ledger-stop` to ask for the slice of the history you want. There are some constraints on start and stop
+offsets which cause PQS to fail-fast if they are violated.
You cannot use:
-1. **Offsets that are outside ledger history**
+1. **Offsets that are outside ledger history**
+
+ .. mermaid::
- ```mermaid
- gantt
+ gantt
title Requested start '00' is outside of ledger history '01...10':
axisFormat %y
Request :, 0000-01-01, 10y
Participant Node :, 0001-01-01, 9y
- ```
- ```mermaid
- gantt
+ .. mermaid::
+
+ gantt
title Requested start '06' is outside of ledger history '01...04':
axisFormat %y
Request :, 0006-01-01, 4y
Participant Node :, 0001-01-01, 3y
- ```
- ```mermaid
- gantt
+ .. mermaid::
+
+ gantt
title Requested end '08' is outside of ledger history '00...03':
axisFormat %y
Request :, 0000-01-01, 8y
Participant Node :, 0000-01-01, 3y
- ```
-2. **Pruned offsets or Genesis on pruned ledger**
+1. **Pruned offsets or Genesis on pruned ledger**
+
+ .. mermaid::
- ```mermaid
- gantt
+ gantt
title Requested start 'GENESIS' is outside of ledger history '02...09':
axisFormat %y
Request :, 0000-01-01, 9y
Participant Node :, 0002-01-01, 7y
- ```
-3. **Offsets that lead to a gap in datastore history**
+1. **Offsets that lead to a gap in datastore history**
- ```mermaid
- gantt
+ .. mermaid::
+
+ gantt
title Requested offsets '05...09' will produce gap in datastore history '01...03':
axisFormat %y
Request :, 0005-01-01, 4y
Datastore :, 0001-01-01, 2y
- ```
-4. **Offsets that are before the PQS datastore history**
+1. **Offsets that are before the PQS datastore history**
+
+ .. mermaid::
- ```mermaid
- gantt
+ gantt
title Cannot prepend to existing datastore. Requested start '00', datastore start '02':
axisFormat %y
Request :, 0000-01-01, 10y
Datastore :, 0002-01-01, 8y
- ```
In the above examples:
- **Request** represents offsets requested via `--pipeline-ledger-start` and `--pipeline-ledger-stop` arguments
+
- **Participant Node** represents the availability of unpruned ledger history in the Participant Node
+
- **Datastore** represents data in the PQS database
### Pruning
-Pruning ledger data from the database can help reduce storage size and improve query performance by removing old and irrelevant data. PQS provides two approaches to prune ledger data: using the PQS CLI or using the `prune_archived_to_offset()` SQL function (see `pqs-references-sql-api`).
+Pruning ledger data from the database can help reduce storage size and improve query performance by removing old and
+irrelevant data. PQS provides two approaches to prune ledger data: using the PQS CLI or using the `prune_archived_to_offset()` SQL
+function (see [SQL API](/appdev/reference/pqs-sql-reference)).
-The legacy `prune_to_offset` function has been deprecated since version 3.5.0. It is known to cause deadlocks and introduces performance bottlenecks.
+The legacy `prune_to_offset` function has been deprecated since version 3.5.0.
+It is known to cause deadlocks and introduces performance bottlenecks.
-It is succeeded by `prune_archived_to_offset`, which resolves these issues by modifying the pruning logic. Unlike the legacy function, the new variant does not collapse historical transactions. Instead, it preserves untouched all active contracts, and maintains their associated events and transactions.
+It is succeeded by `prune_archived_to_offset`, which resolves these issues by modifying the pruning logic.
+Unlike the legacy function, the new variant does not collapse historical transactions.
+Instead, it preserves untouched all active contracts, and maintains their associated events and transactions.
-Decide on what are the oldest offsets that you will ever need in PQS and setup periodic pruning for data at offsets older than that. Thereby ensuring that your query performance does not deteriorate over time as your PQS database continuously increases in size. In case you need all data from ledger begin consider:
+Decide on what are the oldest offsets that you will ever need in PQS and setup periodic pruning for data at offsets
+older than that. Thereby ensuring that your query performance does not deteriorate over time as your PQS database
+continuously increases in size. In case you need all data from ledger begin consider:
- data growth rate, and
+
- size your database server to comfortably hold that data
-Calling either the `prune` CLI command with `--prune-mode Force` or calling the PostgreSQL function `prune_archived_to_offset()` deletes data irrevocably
+Calling either the `prune` CLI command with `--prune-mode Force` or calling the PostgreSQL function
+`prune_archived_to_offset()` deletes data irrevocably
-### Data deletion and changes
+#### Data deletion and changes
Both pruning approaches (CLI and SQL function) share the same behavior in terms of data deletion and changes:
- Removes archived contracts and their associated create/archive events.
+
- Removes exercises and their corresponding exercise events.
+
- Removes transactions that no longer reference active contracts.
They also provide the following guarantees:
- All currently active contracts and their history remain intact.
+
- All data (transactions/events/choices/contracts) for transaction with an offset greater than the pruning target remains intact.
-The target offset, that is, the offset provided via `--prune-target` or as argument to `prune_archived_to_offset()` SQL function is the transaction with the highest offset to be affected by the pruning operation.
+The target offset, that is, the offset provided via `--prune-target` or as argument to `prune_archived_to_offset()` SQL
+function is the transaction with the highest offset to be affected by the pruning operation.
-If the provided offset does not have a transaction associated with it, the effective target offset becomes the oldest offset that succeeds (is greater than) the provided offset.
+If the provided offset does not have a transaction associated with it, the effective target offset becomes the
+oldest offset that succeeds (is greater than) the provided offset.
-Pruning is a destructive operation and cannot be undone. If necessary, make sure to back up your data before performing any pruning operations.
+Pruning is a destructive operation and cannot be undone. If necessary, make sure to back up your data before
+performing any pruning operations.
+
+#### Constraints
-### Constraints
+Some constraints apply to pruning operations (see also [PQS time model](/appdev/reference/pqs-sql-reference#pqs-time-model)):
-Some constraints apply to pruning operations (see also `pqs-references-pqs-time-model`):
+1. The provided target offset must be within the bounds of the contiguous history. If the target offset is outside
+ the bounds, an error is raised.
-1. The provided target offset must be within the bounds of the contiguous history. If the target offset is outside the bounds, an error is raised.
-2. The pruning operation cannot coincide with the latest consistent checkpoint of the contiguous history. If so, it raises an error.
+1. The pruning operation cannot coincide with the latest consistent checkpoint of the contiguous history. If so, it
+ raises an error.
-### Pruning from the command line
+#### Pruning from the command line
-The PQS CLI provides a `prune` command that allows you to prune the ledger data up to a specified offset, timestamp, or duration.
+The PQS CLI provides a `prune` command that allows you to prune the ledger data up to a specified offset,
+timestamp, or duration.
For detailed information on all available options, please run:
-``` text
+```text
$ ./scribe.jar datastore postgres-document prune --help-verbose
```
-To use the `prune` command, you need to provide a pruning target as an argument. The pruning target can be an offset, a timestamp, or a duration (ISO 8601[^1]):
+To use the `prune` command, you need to provide a pruning target as an argument. The pruning target can be an
+offset, a timestamp, or a duration (ISO 8601 [#]_):
-``` text
+```text
$ ./scribe.jar datastore postgres-document prune --prune-target ''
```
-By default, the `prune` command performs a dry run, meaning it displays the effects of the pruning operation without actually deleting any data. To execute the pruning operation, add the `--prune-mode Force` option:
+By default, the `prune` command performs a dry run, meaning it displays the effects of the pruning operation
+without actually deleting any data. To execute the pruning operation, add the `--prune-mode Force` option:
-``` text
+```text
$ ./scribe.jar datastore postgres-document prune --prune-target '' --prune-mode Force
```
-### Example with timestamp and duration
+#### Example with timestamp and duration
-In addition to providing an offset as `--prune-target`, a timestamp or duration can also be used as a pruning cut-off. For example, to prune data older than 30 days (relative to now), you can use the following command:
+In addition to providing an offset as `--prune-target`, a timestamp or duration can also be used as a pruning
+cut-off. For example, to prune data older than 30 days (relative to now), you can use the following command:
-``` text
+```text
$ ./scribe.jar datastore postgres-document prune --prune-target P30D
```
To prune data up to a specific timestamp, you can use the following command:
-``` text
+```text
$ ./scribe.jar datastore postgres-document prune --prune-target 2023-01-30T00:00:00.000Z
```
-### Pruning with sql function
+#### Pruning with sql function
-The `prune_archived_to_offset()` is a SQL function that allows you to prune the ledger data up to a specified offset. It has the same behavior as the `datastore postgres-document prune` command, but does not feature a dry-run option.
+The `prune_archived_to_offset()` is a SQL function that allows you to prune the ledger data up to a specified offset.
+It has the same behavior as the `datastore postgres-document prune` command, but does not feature a dry-run option.
-The legacy `prune_to_offset` function has been deprecated since version 3.5.0. It is known to cause deadlocks and introduces performance bottlenecks.
+The legacy `prune_to_offset` function has been deprecated since version 3.5.0.
+It is known to cause deadlocks and introduces performance bottlenecks.
-It is succeeded by `prune_archived_to_offset`, which resolves these issues by modifying the pruning logic. Unlike the legacy function, the new variant does not collapse historical transactions. Instead, it preserves untouched all active contracts, and maintains their associated events and transactions.
+It is succeeded by `prune_archived_to_offset`, which resolves these issues by modifying the pruning logic.
+Unlike the legacy function, the new variant does not collapse historical transactions.
+Instead, it preserves untouched all active contracts, and maintains their associated events and transactions.
To use `prune_archived_to_offset()`, you need to provide an offset:
@@ -238,7 +276,8 @@ select * from prune_archived_to_offset('');
The function deletes transactions and updates active contracts as described above.
-You can use `prune_archived_to_offset()` in combination with the `nearest_offset()` function to prune data up to a specific timestamp or interval:
+You can use `prune_archived_to_offset()` in combination with the `nearest_offset()` function to prune data up to a specific
+timestamp or interval:
```sql
select * from prune_archived_to_offset(nearest_offset('1970-01-01 08:01:00+08' :: timestamp with time zone));
@@ -246,18 +285,58 @@ select * from prune_archived_to_offset(nearest_offset('PT2H' :: interval));
select * from prune_archived_to_offset(nearest_offset(interval '3 days'));
```
+#### Querying the pruning state
+
+{/* DIVERGENCE: this subsection is 3.5-only and absent from the source COPIED_START block above (docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/operate.rst, hash f00146d1); split into version tabs per PQS 3.4/3.5 divergence, cf-docs PR #1395 follow-up. */}
+
+
+
+
+
+Not available in PQS 3.4. Use the `prune` CLI command's dry-run mode (see [Pruning from the command line](#pruning-from-the-command-line)) to inspect the effect of a pruning operation before applying it.
+
+
+
+
+
+PQS persists the offset of the last pruning operation. You can query it using the `pruned_offset()` function:
+
+```sql
+select pruned_offset();
+```
+
+This returns `NULL` if the database has never been pruned, or the offset of the most recent pruning operation.
+
+Pruning to an offset at or below the last pruned offset is a no-op and returns zeroed statistics.
+
+
+Query functions such as `archives()`, `exercises()`, and related summary functions automatically use the pruned
+offset as their default `from_offset` when available, avoiding unnecessary scans over pruned ranges.
+
+
+
+
+
+
### Resetting
-Reset-to-offset is a manual procedure that deletes all transactions from the PQS database after a given offset. This allows you to restart processing from the offset as if subsequent transactions have never been processed.
+Reset-to-offset is a manual procedure that deletes all transactions from the PQS database after a given offset. This
+allows you to restart processing from the offset as if subsequent transactions have never been processed.
-Reset is a dangerous, destructive, and permanent procedure that needs to be coordinated within the entire ecosystem and not performed in isolation.
+Reset is a dangerous, destructive, and permanent procedure that needs to be coordinated within the entire
+ecosystem and not performed in isolation.
-Reset can be useful to perform a point-in-time rollback of the ledger in a range of circumstances. For example, in the event of:
+Reset can be useful to perform a point-in-time rollback of the ledger in a range of circumstances. For example, in
+the event of:
-1. **Unexepected new entities** - A new scope, such as a Party or template, appears in ledger transactions without coordination. That is, new transactions arrive without ensuring PQS is restarted - to ensure it knows about these new enitities prior.
-2. **Ledger roll-back** - If a ledger is rolled-back due to the disaster recovery process, you will need to perform a similar roll back with PQS. This is a manual process that requires coordination with the Participant Node.
+1. **Unexepected new entities** - A new scope, such as a Party or template, appears in ledger transactions without
+ coordination. That is, new transactions arrive without ensuring PQS is restarted - to ensure it knows about these
+ new enitities prior.
+
+1. **Ledger roll-back** - If a ledger is rolled-back due to the disaster recovery process, you will need to perform a
+ similar roll back with PQS. This is a manual process that requires coordination with the Participant Node.
The procedure:
@@ -277,13 +356,15 @@ The procedure:
where pid <> pg_backend_pid() and datname = current_database();
```
-- Obtain a summary of the scope of the proposed reset and validate that the intended outcome matches your expectations by performing a dry run:
+- Obtain a summary of the scope of the proposed reset and validate that the intended outcome matches your
+ expectations by performing a dry run:
```sql
select * from validate_reset_offset("0000000000000a8000");
```
-- Implement the destructive changes of removing all transactions after the given offset and adjust internal metadata to allow PQS to resume processing from the supplied offset:
+- Implement the destructive changes of removing all transactions after the given offset and adjust internal metadata
+ to allow PQS to resume processing from the supplied offset:
```sql
select * from reset_to_offset("0000000000000a8000");
@@ -295,40 +376,55 @@ The procedure:
- Start PQS.
-- Conduct any remedial action required in PQS database consumers, to account for the fact that the ledger appears to be rolled back to the specified offset.
+- Conduct any remedial action required in PQS database consumers, to account for the fact that the ledger appears to
+ be rolled back to the specified offset.
- Start applications that use the PQS database and resume operation.
-The provided target offset must be within the bounds of the contiguous history. If the target offset is outside the bounds, it raises an error.
+The provided target offset must be within the bounds of the contiguous history. If the target offset is outside the
+bounds, it raises an error.
-If the database was previously pruned and the reset target offset is below the stored pruning offset, `reset_to_offset` retreats the pruning offset to the reset target. After such a reset, `pruned_offset()` returns the new target offset. If the reset target is at or above the pruning offset, the pruning state is preserved.
+If the database was previously pruned and the reset target offset is below the stored pruning offset,
+`reset_to_offset` retreats the pruning offset to the reset target. After such a reset, `pruned_offset()`
+returns the new target offset. If the reset target is at or above the pruning offset, the pruning state is
+preserved.
### Redacting
-The redaction feature enables removal of sensitive or personally identifiable information from contracts and exercises within the PQS database. This operation is particularly useful for complying with privacy regulations and data protection laws, as it enables the permanent removal of contract payloads, contract keys, choice arguments, and choice results. Note that redaction is a destructive operation and once redacted, information cannot be restored.
+The redaction feature enables removal of sensitive or personally identifiable information from contracts and
+exercises within the PQS database. This operation is particularly useful for complying with privacy regulations and
+data protection laws, as it enables the permanent removal of contract payloads, contract keys, choice arguments, and
+choice results. Note that redaction is a destructive operation and once redacted, information cannot be restored.
-The redaction process involves assigning a `redaction_id` to a contract or an exercise and nullifying its sensitive data fields. For contracts, the `payload` and `contract_key` fields are redacted, while for exercises, the `argument` and `result` fields are redacted.
+The redaction process involves assigning a `redaction_id` to a contract or an exercise and nullifying its sensitive
+data fields. For contracts, the `payload` and `contract_key` fields are redacted, while for exercises, the
+`argument` and `result` fields are redacted.
-### Conditions for redaction
+#### Conditions for redaction
The following conditions apply to contracts and interface views:
- You cannot redact an active contract
+
- A redacted contract cannot be redacted again
There are no restrictions on the redaction of choice exercise events.
-A redaction operation requires a redaction ID, which is an arbitrary label to identify the redaction and provide information about its reason, and correlate with other systems that coordinate such activity.
+A redaction operation requires a redaction ID, which is an arbitrary label to identify the redaction and provide
+information about its reason, and correlate with other systems that coordinate such activity.
-### Examples
+#### Examples
-#### Redacting an archived contract
+##### Redacting an archived contract
-To redact an archived contract, use the `redact_contract` function by providing the `contract_id` and a `redaction_id`. The intent of the `redaction_id` is to provide a case reference to identify the reason why the redaction has taken place, and it should be set according to organizational policies. This operation nullifies the `payload` and `contract_key` of the contract and assigns the `redaction_id`.
+To redact an archived contract, use the `redact_contract` function by providing the `contract_id` and a
+`redaction_id`. The intent of the `redaction_id` is to provide a case reference to identify the reason why the
+redaction has taken place, and it should be set according to organizational policies. This operation nullifies the
+`payload` and `contract_key` of the contract and assigns the `redaction_id`.
```sql
select redact_contract('', '');
@@ -336,55 +432,86 @@ select redact_contract('', '');
Redaction is applied to the contract and its interface views, if any, and it returns the number of affected entries.
-#### Redacting a choice exercise
+##### Redacting a choice exercise
-To redact an exercise, use the `redact_exercise` function by providing the `event_id` of the exercise and a `redaction_id`. This nullifies the `argument` and `result` of the exercise and assigns the `redaction_id`.
+To redact an exercise, use the `redact_exercise` function by providing the `event_id` of the exercise and a
+`redaction_id`. This nullifies the `argument` and `result` of the exercise and assigns the `redaction_id`.
```sql
select redact_exercise('', '');
```
-### Accessing redaction information
+#### Accessing redaction information
-The `redaction_id` of a contract is exposed as a column in the following functions of the SQL API. The columns `payload` and `contract_key` for a redacted contract are `NULL`.
+The `redaction_id` of a contract is exposed as a column in the following functions of the SQL API. The columns
+`payload` and `contract_key` for a redacted contract are `NULL`.
- `creates(...)`
+
- `archives(...)`
+
- `active(...)`
+
- `lookup_contract(...)`
-The `redaction_id` of an exercise event is exposed as a column in the following functions of the SQL API. The columns `argument` and `result` for a redacted exercise are `NULL`:
+The `redaction_id` of an exercise event is exposed as a column in the following functions of the SQL API. The
+columns `argument` and `result` for a redacted exercise are `NULL`:
- `exercises(...)`
+
- `lookup_exercises(...)`
-#### Representative package ID support
+[#] https://en.wikipedia.org/wiki/ISO_8601#Durations
+
+{/* COPIED_END */}
+
+{/* DIVERGENCE: "Representative package ID support" existed only in PQS 3.4 (docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/operate.rst, hash f00146d1) and was removed from the 3.5 source; split into version tabs per PQS 3.4/3.5 divergence, cf-docs PR #1395 follow-up. */}
+
+
+
+
+
+##### Representative package ID support
-PQS does not support ingesting contract payloads where the original package ID of a Ledger API create event is not available in the Participant Node's package store that the PQS instance is connected to. Such a situation can occur on an ACS import procedure on the Participant Node, where the original package ID is replaced by its representative package ID
+PQS does not support ingesting contract payloads where the original package ID of a Ledger API create event is not available in the Participant Node's package store that the PQS instance is connected to. Such a situation can occur on an ACS import procedure on the Participant Node, where the original package ID is replaced by its representative package ID.
If you are using ACS import/export procedures that can replace the original package ID of contracts, please ensure that the original package IDs are also uploaded to the Participant Node's package store to avoid disruptions in PQS processing.
-[^1]: [https://en.wikipedia.org/wiki/ISO_8601#Durations](https://en.wikipedia.org/wiki/ISO_8601#Durations)
+
-{/* COPIED_END */}
+
+
+This limitation is no longer documented for PQS 3.5.
+
+
+
+
## Observe
-{/* COPIED_START source="docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/observe.rst" hash="69375e51" */}
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/observe.rst" hash="0266531a" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
-This section describes observability features of PQS, which are designed to help you monitor health and performance of the application.
+This section describes observability features of PQS, which are designed to help you monitor health and performance
+of the application.
### Approach to observability
-PQS opted to incorporate OpenTelemetry APIs to provide its observability features. All three sources of signals (traces, metrics, and logs) can be exported to various backends by providing appropriate configuration defined by OpenTelemetry protocols and guidelines. This makes PQS flexible in terms of observability backends, allowing users to choose what fits their needs and established infrastructure without being overly prescriptive.
+PQS opted to incorporate OpenTelemetry APIs to provide its observability features. All three sources of signals
+(traces, metrics, and logs) can be exported to various backends by providing appropriate configuration defined by
+OpenTelemetry protocols and guidelines. This makes PQS flexible in terms of observability backends, allowing users to
+choose what fits their needs and established infrastructure without being overly prescriptive.
-To have PQS emit observability data, an OpenTelemetry Java Agent must be attached to the JVM running PQS. OpenTelemetry's documentation page on Java Agent Configuration[^1] has all the necessary information to get started.
+To have PQS emit observability data, an OpenTelemetry Java Agent must be attached to the JVM running PQS.
+OpenTelemetry's documentation page on Java Agent Configuration [#]_ has all the necessary information to get started.
-As a frequently requested shortcut (only metrics over Prometheus exposition endpoint embedded by PQS), the following snippet can help you get started. For more details, refer to the official documentation:
+As a frequently requested shortcut (only metrics over Prometheus exposition endpoint embedded by PQS), the following
+snippet can help you get started. For more details, refer to the official documentation:
-``` text
+```text
$ export OTEL_SERVICE_NAME=pqs
$ export OTEL_TRACES_EXPORTER=none
$ export OTEL_LOGS_EXPORTER=none
@@ -394,185 +521,245 @@ $ export JDK_JAVA_OPTIONS="-javaagent:path/to/opentelemetry-javaagent.jar"
$ ./scribe.jar pipeline ledger postgres-document ...
```
-PQS Docker images already come pre-configured this way, but users are free to override these values as they see fit for their environments.
+PQS Docker images already come pre-configured this way, but users are free to override these values as they see fit
+for their environments.
### Logging
-### Log level
+#### Log level
-Set log level with `--logger-level`. Possible value are `All`, `Fatal`, `Error`, `Warning`, `Info` (default), `Debug`, `Trace`:
+Set log level with `--logger-level`. Possible value are `All`, `Fatal`, `Error`, `Warning`, `Info`
+(default), `Debug`, `Trace`:
-``` text
+```text
--logger-level=Debug
```
-### Per-logger log level
+#### Per-logger log level
-Use `--logger-mappings` to adjust the log level for individual loggers. For example, to remove Netty network traffic from a more detailed overall log:
+Use `--logger-mappings` to adjust the log level for individual loggers. For example, to remove Netty network
+traffic from a more detailed overall log:
-``` text
+```text
--logger-mappings-io.netty=Warning \
--logger-mappings-io.grpc.netty=Trace
```
-### Log pattern
+#### Log pattern
-With `--logger-pattern`, use one of the predefined patterns, such as `Plain` (default), `Standard` (standard format used in DA applications), `Structured`, or set your own. Check Log Format Configuration[^2] for more details.
+With `--logger-pattern`, use one of the predefined patterns, such as `Plain` (default), `Standard` (standard
+format used in DA applications), `Structured`, or set your own. Check Log Format Configuration [#]_ for more details.
To use your custom format, provide its string representation, such as:
-``` text
+```text
--logger-pattern="%highlight{%fixed{1}{%level}} [%fiberId] %name:%line %highlight{%message} %highlight{%cause} %kvs"
```
-### Log format for console output
+#### Log format for console output
-Use `--logger-format` to set the log format. Possible values are `Plain` (default) or `Json`. These formats can be used for the `pipeline` command.
+Use `--logger-format` to set the log format. Possible values are `Plain` (default) or `Json`. These formats can
+be used for the `pipeline` command.
-### Log format for file output
+#### Log format for file output
-Use `--logger-format` to set the log format. Possible values are `Plain` (default), `Json`, `PlainAsync` and `JsonAsync`. They can be used for the interactive commands, such as `prune`. For `PlainAsync` and `JsonAsync`, log entries are written to the destination file asynchronously.
+Use `--logger-format` to set the log format. Possible values are `Plain` (default), `Json`, `PlainAsync` and
+`JsonAsync`. They can be used for the interactive commands, such as `prune`. For `PlainAsync` and
+`JsonAsync`, log entries are written to the destination file asynchronously.
-### Destination file for file output
+#### Destination file for file output
-Use `--logger-destination` to set the path to the destination file (default: `output.log`) for interactive commands, such as `prune`.
+Use `--logger-destination` to set the path to the destination file (default: `output.log`) for interactive commands,
+such as `prune`.
-### Log format and log pattern combinations
+#### Log format and log pattern combinations
- `Plain` / `Plain`
- ``` text
- 00:00:23.737 I [zio-fiber-0] com.digitalasset.scribe.pipeline.pipeline.Impl:34 Starting pipeline on behalf of 'Alice_1::12209982174bbaf1e6283234ab828bcab9b73fbe313315b181134bcae9566d3bbf1b' application=scribe
- 00:00:24.658 I [zio-fiber-0] com.digitalasset.scribe.pipeline.pipeline.Impl:61 Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b' application=scribe
- 00:00:25.043 I [zio-fiber-895] com.digitalasset.zio.daml.ledgerapi.package:201 Contract filter inclusive of 1 templates and 0 interfaces application=scribe
- 00:00:25.724 I [zio-fiber-0] com.digitalasset.scribe.pipeline.pipeline.Impl:85 Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b' application=scribe
- ```
+ ```text
+ 00:00:23.737 I [zio-fiber-0] com.digitalasset.scribe.pipeline.pipeline.Impl:34 Starting pipeline on behalf of 'Alice_1::12209982174bbaf1e6283234ab828bcab9b73fbe313315b181134bcae9566d3bbf1b' application=scribe
+ 00:00:24.658 I [zio-fiber-0] com.digitalasset.scribe.pipeline.pipeline.Impl:61 Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b' application=scribe
+ 00:00:25.043 I [zio-fiber-895] com.digitalasset.zio.daml.ledgerapi.package:201 Contract filter inclusive of 1 templates and 0 interfaces application=scribe
+ 00:00:25.724 I [zio-fiber-0] com.digitalasset.scribe.pipeline.pipeline.Impl:85 Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b' application=scribe
+ ```
- `Plain` / `Standard`
- ``` text
- component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:38.902+0000 level=INFO correlation_id=tbd description=Starting pipeline on behalf of 'Alice_1::1220c6d22d46d59c8454bd245e5a3bc238e5024d37bfd843dbad6885674f3a9673c5' scribe=application=scribe
- component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:39.734+0000 level=INFO correlation_id=tbd description=Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b' scribe=application=scribe
- component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:39.982+0000 level=INFO correlation_id=tbd description=Contract filter inclusive of 1 templates and 0 interfaces scribe=application=scribe
- component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:40.476+0000 level=INFO correlation_id=tbd description=Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b' scribe=application=scribe
- ```
+ ```text
+ component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:38.902+0000 level=INFO correlation_id=tbd description=Starting pipeline on behalf of 'Alice_1::1220c6d22d46d59c8454bd245e5a3bc238e5024d37bfd843dbad6885674f3a9673c5' scribe=application=scribe
+ component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:39.734+0000 level=INFO correlation_id=tbd description=Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b' scribe=application=scribe
+ component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:39.982+0000 level=INFO correlation_id=tbd description=Contract filter inclusive of 1 templates and 0 interfaces scribe=application=scribe
+ component=scribe instance_uuid=5f707d27-8188-4a44-904e-2f98ee9f4177 timestamp=2024-01-16T23:42:40.476+0000 level=INFO correlation_id=tbd description=Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b' scribe=application=scribe
+ ```
- `Plain` / `Custom`
- ``` text
- --logger-pattern=%timestamp{yyyy-MM-dd'T'HH:mm:ss} %level %name:%line %highlight{%message} %highlight{%cause} %kvs
- ```
+ ```text
+ --logger-pattern=%timestamp{yyyy-MM-dd'T'HH:mm:ss} %level %name:%line %highlight{%message} %highlight{%cause} %kvs
+ ```
- ``` text
- 2024-01-16T23:55:52 INFO com.digitalasset.scribe.pipeline.pipeline.Impl:34 Starting pipeline on behalf of 'Alice_1::1220444f494b31c0a40c2f393edac3f5900325028c6f810a203a0334cd830ec230c8' application=scribe
- 2024-01-16T23:55:53 INFO com.digitalasset.scribe.pipeline.pipeline.Impl:61 Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b' application=scribe
- 2024-01-16T23:55:53 INFO com.digitalasset.zio.daml.ledgerapi.package:201 Contract filter inclusive of 1 templates and 0 interfaces application=scribe
- 2024-01-16T23:55:53 INFO com.digitalasset.scribe.pipeline.pipeline.Impl:85 Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b' application=scribe
- ```
+ ```text
+ 2024-01-16T23:55:52 INFO com.digitalasset.scribe.pipeline.pipeline.Impl:34 Starting pipeline on behalf of 'Alice_1::1220444f494b31c0a40c2f393edac3f5900325028c6f810a203a0334cd830ec230c8' application=scribe
+ 2024-01-16T23:55:53 INFO com.digitalasset.scribe.pipeline.pipeline.Impl:61 Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b' application=scribe
+ 2024-01-16T23:55:53 INFO com.digitalasset.zio.daml.ledgerapi.package:201 Contract filter inclusive of 1 templates and 0 interfaces application=scribe
+ 2024-01-16T23:55:53 INFO com.digitalasset.scribe.pipeline.pipeline.Impl:85 Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b' application=scribe
+ ```
- `Json` / `Standard`
- ```json
- {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:12.537+0000","level":"INFO","correlation_id":"tbd","description":"Starting pipeline on behalf of 'Alice_1::1220f03ed424480ab4487d88230fc033f3910f4cb4492fea68535a5760744b53dabe'","scribe":{"application":"scribe"}}
- {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:13.551+0000","level":"INFO","correlation_id":"tbd","description":"Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b'","scribe":{"application":"scribe"}}
- {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:13.935+0000","level":"INFO","correlation_id":"tbd","description":"Contract filter inclusive of 1 templates and 0 interfaces","scribe":{"application":"scribe"}}
- {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:14.659+0000","level":"INFO","correlation_id":"tbd","description":"Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b'","scribe":{"application":"scribe"}}
- ```
+ ```json
+ {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:12.537+0000","level":"INFO","correlation_id":"tbd","description":"Starting pipeline on behalf of 'Alice_1::1220f03ed424480ab4487d88230fc033f3910f4cb4492fea68535a5760744b53dabe'","scribe":{"application":"scribe"}}
+ {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:13.551+0000","level":"INFO","correlation_id":"tbd","description":"Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b'","scribe":{"application":"scribe"}}
+ {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:13.935+0000","level":"INFO","correlation_id":"tbd","description":"Contract filter inclusive of 1 templates and 0 interfaces","scribe":{"application":"scribe"}}
+ {"component":"scribe","instance_uuid":"03c263a0-6e3d-416e-b7f2-0e56b9e34841","timestamp":"2024-01-17T00:04:14.659+0000","level":"INFO","correlation_id":"tbd","description":"Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b'","scribe":{"application":"scribe"}}
+ ```
- `Json` / `Structured`
- ```json
- {"timestamp":"2024-01-17T00:08:25+0000","level":"INFO","thread":"zio-fiber-0","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:34","message":"Starting pipeline on behalf of 'Alice_1::122077c6b00e952ff694e2b25b6f5eb9582f815dfe793e2da668b119481a1dd5acdc'","application":"scribe"}
- {"timestamp":"2024-01-17T00:08:26+0000","level":"INFO","thread":"zio-fiber-0","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:61","message":"Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b'","application":"scribe"}
- {"timestamp":"2024-01-17T00:08:26+0000","level":"INFO","thread":"zio-fiber-882","location":"com.digitalasset.zio.daml.ledgerapi.package:201","message":"Contract filter inclusive of 1 templates and 0 interfaces","application":"scribe"}
- {"timestamp":"2024-01-17T00:08:26+0000","level":"INFO","thread":"zio-fiber-0","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:85","message":"Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b'","application":"scribe"}
- ```
+ ```json
+ {"timestamp":"2024-01-17T00:08:25+0000","level":"INFO","thread":"zio-fiber-0","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:34","message":"Starting pipeline on behalf of 'Alice_1::122077c6b00e952ff694e2b25b6f5eb9582f815dfe793e2da668b119481a1dd5acdc'","application":"scribe"}
+ {"timestamp":"2024-01-17T00:08:26+0000","level":"INFO","thread":"zio-fiber-0","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:61","message":"Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b'","application":"scribe"}
+ {"timestamp":"2024-01-17T00:08:26+0000","level":"INFO","thread":"zio-fiber-882","location":"com.digitalasset.zio.daml.ledgerapi.package:201","message":"Contract filter inclusive of 1 templates and 0 interfaces","application":"scribe"}
+ {"timestamp":"2024-01-17T00:08:26+0000","level":"INFO","thread":"zio-fiber-0","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:85","message":"Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b'","application":"scribe"}
+ ```
- `Json` / `Custom`
- ``` text
- --logger-pattern=%label{timestamp}{%timestamp{yyyy-MM-dd'T'HH:mm:ss}} %label{level}{%level} %label{location}{%name:%line} %label{description}{%message} %label{cause}{%cause} %label{scribe}{%kvs}
- ```
+ ```text
+ --logger-pattern=%label{timestamp}{%timestamp{yyyy-MM-dd'T'HH:mm:ss}} %label{level}{%level} %label{location}{%name:%line} %label{description}{%message} %label{cause}{%cause} %label{scribe}{%kvs}
+ ```
- ```json
- {"timestamp":"2024-01-17T00:16:31","level":"INFO","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:34","description":"Starting pipeline on behalf of 'Alice_1::1220ee13431ac437d454ea59d622cfc76599e0846a3caf166b4306d47b1bf83944a6'","scribe":{"application":"scribe"}}
- {"timestamp":"2024-01-17T00:16:33","level":"INFO","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:61","description":"Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b'","scribe":{"application":"scribe"}}
- {"timestamp":"2024-01-17T00:16:34","level":"INFO","location":"com.digitalasset.zio.daml.ledgerapi.package:201","description":"Contract filter inclusive of 1 templates and 0 interfaces","scribe":{"application":"scribe"}}
- {"timestamp":"2024-01-17T00:16:35","level":"INFO","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:85","description":"Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b'","scribe":{"application":"scribe"}}
- ```
+ ```json
+ {"timestamp":"2024-01-17T00:16:31","level":"INFO","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:34","description":"Starting pipeline on behalf of 'Alice_1::1220ee13431ac437d454ea59d622cfc76599e0846a3caf166b4306d47b1bf83944a6'","scribe":{"application":"scribe"}}
+ {"timestamp":"2024-01-17T00:16:33","level":"INFO","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:61","description":"Last checkpoint is absent. Seeding from ACS before processing transactions with starting offset '00000000000000000b'","scribe":{"application":"scribe"}}
+ {"timestamp":"2024-01-17T00:16:34","level":"INFO","location":"com.digitalasset.zio.daml.ledgerapi.package:201","description":"Contract filter inclusive of 1 templates and 0 interfaces","scribe":{"application":"scribe"}}
+ {"timestamp":"2024-01-17T00:16:35","level":"INFO","location":"com.digitalasset.scribe.pipeline.pipeline.Impl:85","description":"Continuing from offset '00000000000000000b' and index '0' until offset '00000000000000000b'","scribe":{"application":"scribe"}}
+ ```
- Notice you need to use `%label{your_label}{format}` to describe a Json attribute-value pair.
+ Notice you need to use `%label{your_label}{format}` to describe a Json attribute-value pair.
### Application metrics
-Assuming PQS exposes metrics as described above, you can access the following metrics at `http://localhost:9090/metrics`. Each metric is accompanied by `# HELP` and `# TYPE` comments, which describe the meaning of the metric and its type, respectively.
+Assuming PQS exposes metrics as described above, you can access the following metrics at
+`http://localhost:9090/metrics`. Each metric is accompanied by `# HELP` and `# TYPE` comments, which describe
+the meaning of the metric and its type, respectively.
-Some metric types have additional constituent parts exposed as separate metrics. For example, a `histogram` metric type tracks `max`, `count`, `sum`, and actual ranged `bucket`s as separate time series. Metrics are labeled where it makes sense, providing additional context such as the type of operation or the template/choice involved.
+Some metric types have additional constituent parts exposed as separate metrics. For example, a `histogram` metric
+type tracks `max`, `count`, `sum`, and actual ranged `bucket`\ s as separate time series. Metrics are labeled
+where it makes sense, providing additional context such as the type of operation or the template/choice involved.
Conceptual list of metrics (refer to actual metric names in the Prometheus output):
-| Type | Name | Description |
-|-----------|---------------------------------------------|-------------------------------------------------------------------------------------------|
-| gauge | `watermark_ix` | Current watermark index (transaction ordinal number for consistent reads) |
-| counter | `pipeline_events_total` | Processed ledger events |
-| histogram | `jdbc_conn_use` | Latency of database connections usage |
-| histogram | `jdbc_conn_isvalid` | Latency of database connection validation |
-| histogram | `jdbc_conn_commit` | Latency of database connection commit |
-| histogram | `total_tx_handling_latency` | Total latency of transaction handling in PQS (observed in LAPI to committed in DB) |
-| gauge | `tx_lag_from_ledger_wallclock` | Lag from ledger (wall-clock delta (in ms) from command completion to receipt by pipeline) |
-| histogram | `pipeline_convert_acs_event` | Latency of converting ACS events |
-| histogram | `pipeline_convert_transaction` | Latency of converting transactions |
-| histogram | `pipeline_prepare_batch_latency` | Latency of preparing batches of statements |
-| histogram | `pipeline_execute_batch_latency` | Latency of executing batches of statements |
-| histogram | `pipeline_progress_watermark_latency` | Latency of watermark progression |
-| histogram | `pipeline_wp_acs_events_size` | Number of in-flight units of work in `pipeline_wp_acs_events` wait point |
-| histogram | `pipeline_wp_acs_statements_size` | Number of in-flight units of work in `pipeline_wp_acs_statements` wait point |
-| histogram | `pipeline_wp_acs_batched_statements_size` | Number of in-flight units of work in `pipeline_wp_acs_batched_statements` wait point |
-| histogram | `pipeline_wp_acs_prepared_statements_size` | Number of in-flight units of work in `pipeline_wp_acs_prepared_statements` wait point |
-| histogram | `pipeline_wp_events_size` | Number of in-flight units of work in `pipeline_wp_events` wait point |
-| histogram | `pipeline_wp_statements_size` | Number of in-flight units of work in `pipeline_wp_statements` wait point |
-| histogram | `pipeline_wp_batched_statements_size` | Number of in-flight units of work in `pipeline_wp_batched_statements` wait point |
-| histogram | `pipeline_wp_prepared_statements_size` | Number of in-flight units of work in `pipeline_wp_prepared_statements` wait point |
-| histogram | `pipeline_wp_watermarks_size` | Number of in-flight units of work in `pipeline_wp_watermarks` wait point |
-| counter | `pipeline_wp_acs_events_total` | Number of units of work processed in `pipeline_wp_acs_events` wait point |
-| counter | `pipeline_wp_acs_statements_total` | Number of units of work processed in `pipeline_wp_acs_statements` wait point |
-| counter | `pipeline_wp_acs_batched_statements_total` | Number of units of work processed in `pipeline_wp_acs_batched_statements` wait point |
-| counter | `pipeline_wp_acs_prepared_statements_total` | Number of units of work processed in `pipeline_wp_acs_prepared_statements` wait point |
-| counter | `pipeline_wp_events_total` | Number of units of work processed in `pipeline_wp_events` wait point |
-| counter | `pipeline_wp_statements_total` | Number of units of work processed in `pipeline_wp_statements` wait point |
-| counter | `pipeline_wp_batched_statements_total` | Number of units of work processed in `pipeline_wp_batched_statements` wait point |
-| counter | `pipeline_wp_prepared_statements_total` | Number of units of work processed in `pipeline_wp_prepared_statements` wait point |
-| counter | `pipeline_wp_watermarks_total` | Number of units of work processed in `pipeline_wp_watermarks` wait point |
-| counter | `app_restarts_total` | Tracks number of times recoverable failures forced the pipeline to restart |
-| gauge | `grpc_up` | Indicator whether gRPC channel is up and operational |
-| gauge | `jdbc_conn_pool_up` | Indicator whether JDBC connection pool is up and operational |
-
-### Grafana dashboard
-
-Based on the metrics described above, it is possible to build a comprehensive dashboard to monitor PQS. Vendor-supplied Grafana dashboard for PQS can be downloaded from artifacts repository (see `pqs-download`). You may want to refer to this as a starting point for your own.
-
-``` text
+| Type | Name | Description |
+| --- | --- | --- |
+| gauge | `watermark_ix` | Current watermark index (transaction ordinal number for consistent reads) |
+| counter | `pipeline_events_total` | Processed ledger events |
+| histogram | `jdbc_conn_use` | Latency of database connections usage |
+| histogram | `jdbc_conn_isvalid` | Latency of database connection validation |
+| histogram | `jdbc_conn_commit` | Latency of database connection commit |
+| histogram | `total_tx_handling_latency` | Total latency of transaction handling in PQS (observed in LAPI to committed in DB) |
+| gauge | `tx_lag_from_ledger_wallclock` | Lag from ledger (wall-clock delta (in ms) from command completion to receipt by pipeline) |
+| histogram | `pipeline_convert_acs_event` | Latency of converting ACS events |
+| histogram | `pipeline_convert_transaction` | Latency of converting transactions |
+| histogram | `pipeline_prepare_batch_latency` | Latency of preparing batches of statements |
+| histogram | `pipeline_execute_batch_latency` | Latency of executing batches of statements |
+| histogram | `pipeline_progress_watermark_latency` | Latency of watermark progression |
+| histogram | `pipeline_wp_acs_events_size` | Number of in-flight units of work in `pipeline_wp_acs_events` wait point |
+| histogram | `pipeline_wp_acs_statements_size` | Number of in-flight units of work in `pipeline_wp_acs_statements` wait point |
+| histogram | `pipeline_wp_acs_batched_statements_size` | Number of in-flight units of work in `pipeline_wp_acs_batched_statements` wait point |
+| histogram | `pipeline_wp_acs_prepared_statements_size` | Number of in-flight units of work in `pipeline_wp_acs_prepared_statements` wait point |
+| histogram | `pipeline_wp_events_size` | Number of in-flight units of work in `pipeline_wp_events` wait point |
+| histogram | `pipeline_wp_statements_size` | Number of in-flight units of work in `pipeline_wp_statements` wait point |
+| histogram | `pipeline_wp_batched_statements_size` | Number of in-flight units of work in `pipeline_wp_batched_statements` wait point |
+| histogram | `pipeline_wp_prepared_statements_size` | Number of in-flight units of work in `pipeline_wp_prepared_statements` wait point |
+| histogram | `pipeline_wp_watermarks_size` | Number of in-flight units of work in `pipeline_wp_watermarks` wait point |
+| counter | `pipeline_wp_acs_events_total` | Number of units of work processed in `pipeline_wp_acs_events` wait point |
+| counter | `pipeline_wp_acs_statements_total` | Number of units of work processed in `pipeline_wp_acs_statements` wait point |
+| counter | `pipeline_wp_acs_batched_statements_total` | Number of units of work processed in `pipeline_wp_acs_batched_statements` wait point |
+| counter | `pipeline_wp_acs_prepared_statements_total` | Number of units of work processed in `pipeline_wp_acs_prepared_statements` wait point |
+| counter | `pipeline_wp_events_total` | Number of units of work processed in `pipeline_wp_events` wait point |
+| counter | `pipeline_wp_statements_total` | Number of units of work processed in `pipeline_wp_statements` wait point |
+| counter | `pipeline_wp_batched_statements_total` | Number of units of work processed in `pipeline_wp_batched_statements` wait point |
+| counter | `pipeline_wp_prepared_statements_total` | Number of units of work processed in `pipeline_wp_prepared_statements` wait point |
+| counter | `pipeline_wp_watermarks_total` | Number of units of work processed in `pipeline_wp_watermarks` wait point |
+| counter | `app_restarts_total` | Tracks number of times recoverable failures forced the pipeline to restart. Labeled with `exception` indicating the failure category: `"Recoverable GRPC exception."`, `"Recoverable JDBC exception."`, `"Recoverable IO Exception."`, `"Recoverable: unknown Daml package detected."` |
+| gauge | `grpc_up` | Indicator whether gRPC channel is up and operational |
+| gauge | `jdbc_conn_pool_up` | Indicator whether JDBC connection pool is up and operational |
+| gauge | `stream_up` | Indicator whether the pipeline stream is running (including seeding from ACS if requested) |
+
+#### Grafana dashboard
+
+Based on the metrics described above, it is possible to build a comprehensive dashboard to monitor PQS.
+Vendor-supplied Grafana dashboard for PQS can be downloaded from artifacts repository (see [Download](/sdks-tools/development-tools/pqs#install-and-download)). You
+may want to refer to this as a starting point for your own.
+
+```text
grafana/v9.4.0/dashboard.json
grafana/v10.4.0/dashboard.json
grafana/v11.0.0/dashboard.json
```
-
+
### Health check
-The health of the PQS process can be monitored using the health check endpoint `/livez`. The health check endpoint is available on the configured network interface (`--health-address`) and TCP port (`--health-port`). Note the default is `127.0.0.1:8080`.
+PQS exposes health check endpoints on the configured network interface (`--health-address`) and TCP port
+(`--health-port`). The default is `127.0.0.1:8080`.
-``` text
+#### Liveness: `/livez`
+
+Returns HTTP 200 when the PQS process is running. This endpoint is suitable for container liveness probes.
+
+```text
$ curl http://localhost:8080/livez
{"status":"ok"}
```
+#### Readiness: `/readyz`
+
+{/* DIVERGENCE: `/readyz` is 3.5-only and absent from the source COPIED_START block above (docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/observe.rst, hash 69375e51); split into version tabs per PQS 3.4/3.5 divergence, cf-docs PR #1395 follow-up. */}
+
+
+
+
+
+Not available in PQS 3.4. Use `/livez` for liveness probes; there is no separate readiness endpoint.
+
+
+
+
+
+Returns HTTP 200 when PQS can reach both the Participant Node (gRPC) and PostgreSQL (JDBC). Returns HTTP 503
+otherwise.
+
+The response body includes the status of each connectivity check, plus the `stream_up` field indicating whether the
+pipeline is actively processing. Note that `stream_up` is informational only and does not affect the HTTP status
+code. It can briefly be `false` during normal operations such as Daml package reloads.
+
+```text
+$ curl http://localhost:8080/readyz
+{"status":"ok","grpc_up":true,"jdbc_connection_pool_up":true,"stream_up":true}
+```
+
+When a backend is unreachable:
+
+```text
+$ curl -w '\n%{http_code}' http://localhost:8080/readyz
+{"status":"unavailable","grpc_up":false,"jdbc_connection_pool_up":true,"stream_up":false}
+503
+```
+
+
+
+
+
### Tracing of pipeline execution
-PQS instruments the most critical parts of its operations with tracing to provide insights into the execution flow and performance. Traces can be exported to various OpenTelemetry backends by providing appropriate configuration, for example:
+PQS instruments the most critical parts of its operations with tracing to provide insights into the execution flow
+and performance. Traces can be exported to various OpenTelemetry backends by providing appropriate configuration, for
+example:
-``` text
+```text
$ export OTEL_TRACES_EXPORTER=otlp
$ export OTEL_EXPORTER_OTLP_PROTOCOL=grpc
$ export OTEL_EXPORTER_OTLP_ENDPOINT="http://otel-collector:4317"
@@ -582,22 +769,27 @@ $ ./scribe.jar pipeline ledger postgres-document ...
The following root spans are emitted by PQS:
-| span name | description |
-|---------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------|
-| `process metadata and schema` | interactions that happen when PQS starts up and ensures its datastore is ready for operations |
-| `initialization routine` | interactions that happen when PQS establishes its offset range boundaries (including seeding from ACS if requested) on startup |
-| `consume com.daml.ledger.api.v1.TransactionService/GetTransactions` `consume com.daml.ledger.api.v1.TransactionService/GetTransactionTrees` | **\[Daml SDK v2.x\]** timeline of processing a ledger transaction from delivery over gRPC to its persistence to datastore |
-| `consume com.daml.ledger.api.v2.UpdateService/GetUpdates` `consume com.daml.ledger.api.v2.UpdateService/GetUpdateTrees` | **\[Daml SDK v3.x\]** timeline of processing a ledger transaction from delivery over gRPC to its persistence to datastore |
-| `execute datastore transaction` | interactions when a batch of transactions is persisted to the datastore |
-| `advance datastore watermark` | interactions when the latest consecutive watermark is persisted to the datastore |
+| span name | description |
+| --- | --- |
+| `process metadata and schema` | interactions that happen when PQS starts up and ensures its datastore is ready for operations |
+| `initialization routine` | interactions that happen when PQS establishes its offset range boundaries (including seeding from ACS if requested) on startup |
+| `consume com.daml.ledger.api.v1.TransactionService/GetTransactions` `consume com.daml.ledger.api.v1.TransactionService/GetTransactionTrees` | **[Daml SDK v2.x]** timeline of processing a ledger transaction from delivery over gRPC to its persistence to datastore |
+| `consume com.daml.ledger.api.v2.UpdateService/GetUpdates` `consume com.daml.ledger.api.v2.UpdateService/GetUpdateTrees` | **[Daml SDK v3.x]** timeline of processing a ledger transaction from delivery over gRPC to its persistence to datastore |
+| `execute datastore transaction` | interactions when a batch of transactions is persisted to the datastore |
+| `advance datastore watermark` | interactions when the latest consecutive watermark is persisted to the datastore |
-All spans are enriched with contextual information through OpenTelemetry's attributes and events where appropriate. It is advisable to get to know this contextual data. Due to the technical nature of asynchronous and parallel execution, PQS heavily employs span links[^3] to highlight causal relationships between independent traces. Modern trace visualisation tools leverage this information to provide a usable representation and navigation through the involved traces.
+All spans are enriched with contextual information through OpenTelemetry's attributes and events where appropriate.
+It is advisable to get to know this contextual data. Due to the technical nature of asynchronous and parallel
+execution, PQS heavily employs span links [#]_ to highlight causal relationships between independent traces. Modern
+trace visualisation tools leverage this information to provide a usable representation and navigation through the
+involved traces.
-Below is an example of causal trace data that spans receipt of a transaction from the Ledger API all the way to it becoming visible by PQS' SQL API in Postgres.
+Below is an example of causal trace data that spans receipt of a transaction from the Ledger API all the way to it
+becoming visible by PQS' SQL API in Postgres.
-
+
-``` text
+```text
Span #110
Trace ID : 042ce1ffa24b34b38472933ac8209d54
Parent ID :
@@ -681,9 +873,9 @@ SpanLink #2
-> target: Str(↧ advance watermark)
```
-
+
-``` text
+```text
Span #115
Trace ID : 76c58361d46c08761c37ef5821e8fb78
Parent ID :
@@ -791,9 +983,9 @@ Status code : Unset
Status message :
```
-
+
-``` text
+```text
Span #124
Trace ID : 71e67e2420deeef36ef3efacea6399dc
Parent ID :
@@ -861,7 +1053,12 @@ Status message :
### Trace context propagation
-PQS is an intermediary between a ledger instance and downstream applications that would prefer to access data through SQL rather than in streaming manner from Ledger API directly. Despite forming a pipeline between two data storage systems (Canton and PostgreSQL), PQS stores the original ledger transaction's trace context (see also `open-tracing-ledger-api-client`) for the purposes of propagation rather than its own. This allows downstream applications to decide for themselves how they want to connect to the original submission's trace (as a child span or as a new trace connected through span links).
+PQS is an intermediary between a ledger instance and downstream applications that would prefer to access data through
+SQL rather than in streaming manner from Ledger API directly. Despite forming a pipeline between two data storage
+systems (Canton and PostgreSQL), PQS stores the original ledger transaction's trace context
+(see also [Open Tracing in Ledger API Client Applications](/appdev/deep-dives/open-tracing)) for the purposes of propagation rather than its own. This
+allows downstream applications to decide for themselves how they want to connect to the original submission's trace
+(as a child span or as a new trace connected through span links).
```sql
select "offset",
@@ -870,13 +1067,13 @@ select "offset",
from transactions limit 1;
```
-``` text
-offset | trace_parent | trace_state
+```text
+ offset | trace_parent | trace_state
--------------------+---------------------------------------------------------+-----------------
-0000000000000000bb | 00-f35923baa38cc520a1fc3aec6771380b-b4cf363cbf5efa6a-01 | foo=bar,baz=qux
+ 0000000000000000bb | 00-f35923baa38cc520a1fc3aec6771380b-b4cf363cbf5efa6a-01 | foo=bar,baz=qux
```
-``` text
+```text
Span #85
Trace ID : f35923baa38cc520a1fc3aec6771380b
Parent ID : d3300bedd4c64511
@@ -916,18 +1113,21 @@ Attributes:
-> rpc.grpc.status_code: Int(0)
```
-Accessing data stored in PQS' `transactions.trace_context` column allows any application to re-create the propagated trace context[^4] and use it with their runtime's instrumentation library.
+Accessing data stored in PQS' `transactions.trace_context` column allows any application to re-create the
+propagated trace context [#]_ and use it with their runtime's instrumentation library.
### Diagnostics
-PQS is capable of exporting diagnostic telemetry snapshots. This data export archive contains essential troubleshooting information such as:
+PQS is capable of exporting diagnostic telemetry snapshots. This data export archive contains essential
+troubleshooting information such as:
- application thread dumps (over a period of time)
+
- application metrics (over a period of time)
Getting this archive is as easy as accessing the socket with `netcat` tool:
-``` text
+```text
$ nc localhost 9091 > health-dump.zip
$ unzip health-dump.zip
Archive: health-dump.zip
@@ -937,100 +1137,140 @@ Archive: health-dump.zip
The table below lists the available configuration sources with priority decreasing from left to right:
-| System property | Environment variable | Default value | Description |
-|--------------------------------------|--------------------------------------|---------------|-------------------------------------------------------------------------------------------------------------------------------------------------|
-| `da.diagnostics.enabled` | `DA_DIAGNOSTICS_ENABLED` | `true` | Enables/disables diagnostics data collection and exposition |
-| `da.diagnostics.host` | `DA_DIAGNOSTICS_HOST` | `127.0.0.1` | Hostname or IP address to use for binding the exposition socket |
-| `da.diagnostics.port` | `DA_DIAGNOSTICS_PORT` | `0` | Port to use for binding the exposition socket (`0` = random port) |
-| `da.diagnostics.dump.path` | `DA_DIAGNOSTICS_DUMP_PATH` | `` | Directory to write to on graceful shutdown (path needs to be an existing writable directory) |
-| `da.diagnostics.metrics.interval` | `DA_DIAGNOSTICS_METRICS_INTERVAL` | `PT10S` | Metrics collection interval in ISO 8601 format |
-| `da.diagnostics.metrics.buffer.size` | `DA_DIAGNOSTICS_METRICS_BUFFER_SIZE` | `60` | Quantity of samples to store for each monitored metric **(rolling window)** |
-| `da.diagnostics.metrics.tags` | `DA_DIAGNOSTICS_METRICS_TAGS` | `` | Comma-separated list of additional labels to enrich each metric with during exposition (for example, `job=myapp,env=staging,deployed=20250101`) |
-| `da.diagnostics.threads.interval` | `DA_DIAGNOSTICS_THREADS_INTERVAL` | `PT1M` | Thread dumps collection interval in ISO 8601 format |
-| `da.diagnostics.threads.buffer.size` | `DA_DIAGNOSTICS_THREADS_BUFFER_SIZE` | `10` | Quantity of thread dumps to store **(rolling window)** |
-
-[^1]: [https://opentelemetry.io/docs/zero-code/java/agent/configuration/](https://opentelemetry.io/docs/zero-code/java/agent/configuration/)
-
-[^2]: [https://zio.dev/zio-logging/formatting-log-records/#log-format-configuration](https://zio.dev/zio-logging/formatting-log-records/#log-format-configuration)
-
-[^3]: [https://opentelemetry.io/docs/specs/otel/overview/#links-between-spans](https://opentelemetry.io/docs/specs/otel/overview/#links-between-spans)
-
-[^4]: [https://www.w3.org/TR/trace-context/](https://www.w3.org/TR/trace-context/)
+| System property | Environment variable | Default value | Description |
+| --- | --- | --- | --- |
+| `da.diagnostics.enabled` | `DA_DIAGNOSTICS_ENABLED` | `true` | Enables/disables diagnostics data collection and exposition |
+| `da.diagnostics.host` | `DA_DIAGNOSTICS_HOST` | `127.0.0.1` | Hostname or IP address to use for binding the exposition socket |
+| `da.diagnostics.port` | `DA_DIAGNOSTICS_PORT` | `0` | Port to use for binding the exposition socket (`0` = random port) |
+| `da.diagnostics.dump.path` | `DA_DIAGNOSTICS_DUMP_PATH` | `\` | Directory to write to on graceful shutdown (path needs to be an existing writable directory) |
+| `da.diagnostics.metrics.interval` | `DA_DIAGNOSTICS_METRICS_INTERVAL` | `PT10S` | Metrics collection interval in ISO 8601 format |
+| `da.diagnostics.metrics.buffer.size` | `DA_DIAGNOSTICS_METRICS_BUFFER_SIZE` | `60` | Quantity of samples to store for each monitored metric **(rolling window)** |
+| `da.diagnostics.metrics.tags` | `DA_DIAGNOSTICS_METRICS_TAGS` | `\` | Comma-separated list of additional labels to enrich each metric with during exposition (for example, `job=myapp,env=staging,deployed=20250101`) |
+| `da.diagnostics.threads.interval` | `DA_DIAGNOSTICS_THREADS_INTERVAL` | `PT1M` | Thread dumps collection interval in ISO 8601 format |
+| `da.diagnostics.threads.buffer.size` | `DA_DIAGNOSTICS_THREADS_BUFFER_SIZE` | `10` | Quantity of thread dumps to store **(rolling window)** |
+
+[#] https://opentelemetry.io/docs/zero-code/java/agent/configuration/
+[#] https://zio.dev/zio-logging/formatting-log-records/#log-format-configuration
+[#] https://opentelemetry.io/docs/specs/otel/overview/#links-between-spans
+[#] https://www.w3.org/TR/trace-context/
{/* COPIED_END */}
## Recover
-{/* COPIED_START source="docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/recover.rst" hash="0a28a79f" */}
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/recover.rst" hash="65e938ee" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
PQS is designed to operate as a long-running process which uses these principles to enhance availability:
-- **Redundancy** involves running multiple instances of PQS in parallel to ensure that the system remains available even if one instance fails.
-- **Retry** involves healing from transient and recoverable failures without shutting down the process or requiring operator intervention.
-- **Recovery** entails reconciling the current state of the ledger with already exported data in the datastore after a cold start, and continuing from the latest checkpoint.
+- **Redundancy** involves running multiple instances of PQS in parallel to ensure that the system remains available
+ even if one instance fails.
-### High availability
+- **Retry** involves healing from transient and recoverable failures without shutting down the process or requiring
+ operator intervention.
-Multiple isolated instances of PQS can be instantiated without any cross-dependency. This allows for an active-active high availability clustering model. Please note that different instances might not be at the same offset due to different processing rates and general network non-determinism. PQS' SQL API provides capabilities to deal with this 'eventual consistency' model, to ensure that readers have at least 'repeatable read' consistency. See `validate_offset_exists()` in `pqs-references-offset-management` for more details.
+- **Recovery** entails reconciling the current state of the ledger with already exported data in the datastore after
+ a cold start, and continuing from the latest checkpoint.
-```mermaid
----
-title: High Availability Deployment
----
-flowchart LR
- Participant --Ledger API--> PQS1[PQS
Process]
- Participant --Ledger API--> PQS2[PQS
Process]
- PQS1 --JDBC--> Database1[PQS
Database]
- PQS2 --JDBC--> Database2[PQS
Database]
- Database1 <--JDBC--> LoadBalancer[Load
Balancer]
- Database2 <--JDBC--> LoadBalancer[Load
Balancer]
- LoadBalancer <--JDBC--> App((App
Cluster))
- style PQS1 stroke-width:4px
- style PQS2 stroke-width:4px
- style Database1 stroke-width:4px
- style Database2 stroke-width:4px
-```
+### High availability
+
+Multiple isolated instances of PQS can be instantiated without any cross-dependency. This allows for an active-active
+high availability clustering model. Please note that different instances might not be at the same offset due to
+different processing rates and general network non-determinism. PQS' SQL API provides capabilities to deal with
+this 'eventual consistency' model, to ensure that readers have at least 'repeatable read' consistency. See
+`validate_offset_exists()` in [Offset management](/appdev/reference/pqs-sql-reference#offset-management) for more details.
+
+.. mermaid::
+
+ ---
+ title: High Availability Deployment
+ ---
+ flowchart LR
+ Participant --Ledger API--> PQS1[PQS\
Process]
+ Participant --Ledger API--> PQS2[PQS\
Process]
+ PQS1 --JDBC--> Database1[PQS\
Database]
+ PQS2 --JDBC--> Database2[PQS\
Database]
+ Database1 \<--JDBC--> LoadBalancer[Load\
Balancer]
+ Database2 \<--JDBC--> LoadBalancer[Load\
Balancer]
+ LoadBalancer \<--JDBC--> App((App\
Cluster))
+ style PQS1 stroke-width:4px
+ style PQS2 stroke-width:4px
+ style Database1 stroke-width:4px
+ style Database2 stroke-width:4px
### Retries
-PQS' `pipeline` command is a unidirectional streaming process that heavily relies on the availability of its `source` and `target` dependencies. When PQS encounters an error, it attempts to recover by restarting its internal engine, if the error is designated as recoverable:
+PQS' `pipeline` command is a unidirectional streaming process that heavily relies on the availability of its
+`source` and `target` dependencies. When PQS encounters an error, it attempts to recover by restarting its
+internal engine, if the error is designated as recoverable:
+
+- gRPC [#]_ (white-listed; retries if):
-- gRPC[^1] (white-listed; retries if):
- `CANCELLED`
+
- `DEADLINE_EXCEEDED`
+
- `NOT_FOUND`
+
- `PERMISSION_DENIED`
+
- `RESOURCE_EXHAUSTED`
+
- `FAILED_PRECONDITION`
+
- `ABORTED`
+
- `INTERNAL`
+
- `UNAVAILABLE`
+
- `DATA_LOSS`
+
- `UNAUTHENTICATED`
-- JDBC[^2] (black-listed; retries unless):
+
+- JDBC [#]_ (black-listed; retries unless):
+
- `INVALID_PARAMETER_TYPE`
+
- `PROTOCOL_VIOLATION`
+
- `NOT_IMPLEMENTED`
+
- `INVALID_PARAMETER_VALUE`
+
- `SYNTAX_ERROR`
+
- `UNDEFINED_COLUMN`
+
- `UNDEFINED_OBJECT`
+
- `UNDEFINED_TABLE`
+
- `UNDEFINED_FUNCTION`
+
- `NUMERIC_CONSTANT_OUT_OF_RANGE`
+
- `NUMERIC_VALUE_OUT_OF_RANGE`
+
- `DATA_TYPE_MISMATCH`
+
- `INVALID_NAME`
+
- `CANNOT_COERCE`
+
- `UNEXPECTED_ERROR`
+
- Daml packages (retries if):
+
- A transaction references a Daml package that was not present when the pipeline started.
+ See [Dynamic Daml package reload](#dynamic-daml-package-reload).
-### Configuration
+#### Configuration
-The following `pqs-references-configuration-options` are available to control the retry behavior of PQS:
+The following *Configuration options* are available to control the retry behavior of PQS:
-``` text
+```text
--retry-backoff-base string Base time (ISO 8601) for backoff retry strategy (default: PT1S)
--retry-backoff-cap string Max duration (ISO 8601) between attempts (default: PT1M)
--retry-backoff-factor double Factor for backoff retry strategy (default: 2.0)
@@ -1041,38 +1281,17 @@ The following `pqs-references-configuration-options` are available to control th
Configuring `--retry-backoff-*` settings control periodicity of retries and the maximum duration between attempts.
-Configuring `--retry-counter-attempts` and `--retry-counter-duration` controls the maximum *instability* tolerance before shutting down.
+Configuring `--retry-counter-attempts` and `--retry-counter-duration` controls the maximum *instability*
+tolerance before shutting down.
-Configuring `--retry-counter-reset` controls the period of *stability* after which the retry counters are reset across the board.
+Configuring `--retry-counter-reset` controls the period of *stability* after which the retry counters are reset
+across the board.
-### Dynamic Daml package reload
-
-When PQS encounters a transaction referencing an unknown Daml package, it does not terminate. Instead:
-
-1. Ingestion pauses.
-2. PQS fetches new packages from the Participant Node.
-3. Type mappings are rebuilt.
-4. Ingestion resumes from the paused offset.
-
-This handles the case where new DARs are deployed to the Participant Node while PQS is running. No restart is needed.
-
-**Log messages** to look for:
-
-```
-Recoverable: unknown Daml package detected.
-```
-
-**Metric**: `app_restarts_total` with label `exception="Recoverable: unknown Daml package detected."` increments each time this occurs.
-
-
-Dynamic package reload only applies to newly deployed Daml packages. Other configuration changes (parties, templates, interfaces in filters) still require a restart.
-
-
-### Logging
+#### Logging
While PQS recovers, the following log messages are emitted to indicate the progress of the recovery:
-``` text
+```text
12:52:26.753 I [zio-fiber-257] com.digitalasset.scribe.appversion.package:14 scribe, version: UNSPECIFIED application=scribe
12:52:16.725 I [zio-fiber-0] com.digitalasset.scribe.pipeline.Retry.retryRecoverable:48 Recoverable GRPC exception. Attempt 1, unstable for 0 seconds. Remaining attempts: 42. Remaining time: 10 minutes. Exception in thread "zio-fiber-" java.lang.Throwable: Recoverable GRPC exception.
Suppressed: io.grpc.StatusException: UNAVAILABLE: io exception
@@ -1095,31 +1314,38 @@ While PQS recovers, the following log messages are emitted to indicate the progr
Suppressed: java.net.ConnectException: Connection refused application=scribe
```
-### Metrics
+#### Metrics
-The following metrics are available to monitor stability of PQS' dependencies. See `pqs-application-metrics` for more details on general observability:
+The following metrics are available to monitor stability of PQS' dependencies. See [Application metrics](#application-metrics)
+for more details on general observability:
-``` text
-### TYPE app_restarts_total counter
-### HELP app_restarts_total Number of total app restarts due to recoverable errors
+```text
+## TYPE app_restarts_total counter
+## HELP app_restarts_total Number of total app restarts due to recoverable errors
app_restarts_total{,exception="Recoverable GRPC exception."} 5.0
+app_restarts_total{,exception="Recoverable JDBC exception."} 1.0
+app_restarts_total{,exception="Recoverable: unknown Daml package detected."} 1.0
-### TYPE grpc_up gauge
-### HELP grpc_up Grpc channel is up
+## TYPE grpc_up gauge
+## HELP grpc_up Grpc channel is up
grpc_up{} 1.0
-### TYPE jdbc_conn_pool_up gauge
-### HELP jdbc_conn_pool_up JDBC connection pool is up
+## TYPE jdbc_conn_pool_up gauge
+## HELP jdbc_conn_pool_up JDBC connection pool is up
jdbc_conn_pool_up{} 1.0
```
-### Retry counters reset
+#### Retry counters reset
-If PQS encounters network unavailability it starts incrementing retry counters with each attempt. These counters are reset only after a period of stability, as defined by `--retry-counter-reset`. As such, during the prolonged periods of intermittent failures that alternate with brief periods of operating normally, PQS keeps maintaining a cautious stance on assumptions regarding the stability of the overall system. This can be illustrated with an example below:
+If PQS encounters network unavailability it starts incrementing retry counters with each attempt. These counters are
+reset only after a period of stability, as defined by `--retry-counter-reset`. As such, during the prolonged
+periods of intermittent failures that alternate with brief periods of operating normally, PQS keeps maintaining a
+cautious stance on assumptions regarding the stability of the overall system. This can be illustrated with an example
+below:
As an example, for the setting `--retry-counter-reset PT5M` the following timeline illustrates how the retry works:
-``` text
+```text
time --> 1:00 5:00 10:00
v v v
operation: ====xx=x====x=======x========================
@@ -1130,66 +1356,153 @@ x - a failure causing retry happens
= - operating normally
```
-In the timeline above, intermittent failures start at point A, and each retry attempt contributes to the increase of the overall backoff schedule. Consequently, each subsequent retry allows more time for the system to recover. This schedule does not reset to its initial values until after the configured period of stability is reached following the last failure (point B), such as after operating without any failures for 5 minutes (point C).
+In the timeline above, intermittent failures start at point A, and each retry attempt contributes to the increase of
+the overall backoff schedule. Consequently, each subsequent retry allows more time for the system to recover. This
+schedule does not reset to its initial values until after the configured period of stability is reached following the
+last failure (point B), such as after operating without any failures for 5 minutes (point C).
+
+#### Dynamic Daml package reload
+
+{/* DIVERGENCE: this subsection's body differs substantively between the 3.4 and 3.5 sources (docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/recover.rst hash 0a28a79f vs .../3.5/.../recover.rst hash 65e938ee) -- not just reflow; split into version tabs per PQS 3.4/3.5 divergence, cf-docs PR #1395 follow-up. */}
+
+
+
+
+
+When PQS encounters a transaction referencing an unknown Daml package, it does not terminate. Instead:
+
+1. Ingestion pauses.
+2. PQS fetches new packages from the Participant Node.
+3. Type mappings are rebuilt.
+4. Ingestion resumes from the paused offset.
+
+This handles the case where new DARs are deployed to the Participant Node while PQS is running. No restart is needed.
+
+**Log messages** to look for:
+
+```
+Recoverable: unknown Daml package detected.
+```
+
+**Metric**: `app_restarts_total` with label `exception="Recoverable: unknown Daml package detected."` increments each time this occurs.
+
+
+Dynamic package reload only applies to newly deployed Daml packages. Other configuration changes (parties, templates, interfaces in filters) still require a restart.
+
+
+
+
+
+
+Deploying new Daml packages (DARs) to the Participant Node does not require restarting PQS. When PQS encounters an unknown package while processing a received event, it fetches the missing packages from the Participant Node.
+
+This applies when PQS receives a transaction that references a Daml package that was not present when the current
+pipeline session started. PQS temporarily pauses ingestion while it reloads its internal schema:
+
+1. Fetches the updated set of packages from the Participant Node
+1. Parses the new package and rebuilds internal type mappings
+1. Registers new templates and choices in the datastore
+1. Resumes ingestion from the last committed checkpoint
+
+No data is lost during this process. The transaction that triggered the reload is reprocessed after the schema update
+completes.
+
+The following log message indicates that a reload was triggered:
+
+```text
+12:34:56.789 I [zio-fiber-0] com.digitalasset.scribe.pipeline.Retry.retryRecoverable:48 Recoverable: unknown Daml package detected. Attempt 1, unstable for 0 seconds. Remaining attempts: 42. Remaining time: 10 minutes. Exception in thread "zio-fiber-" java.lang.Throwable: Recoverable: unknown Daml package detected.
+ Suppressed: com.digitalasset.zio.daml.ledgerapi.UnknownDamlPackageException: No package for abc123:My.Module:MyTemplate was seen on initialization. Retrying to discover new packages.
+```
+
+The key fragments to search for in logs:
+
+- `Recoverable: unknown Daml package detected`---indicates PQS identified a new package and is reloading.
+
+- `No package for \:\:\ was seen on initialization. Retrying to discover new packages.`
+ ---identifies the specific template from the unrecognized package that triggered the reload.
+
+Reload duration depends on the total number of packages on the ledger and their complexity. For ledgers with many
+packages (100+), the reload may take tens of seconds while PQS re-parses all packages to rebuild its type
+information. During this time, ingestion is paused and downstream queries continue to serve previously committed data.
+
+The `app_restarts_total` counter (see [Application metrics](#application-metrics)) is incremented with label
+`exception="Recoverable: unknown Daml package detected."` each time a reload occurs. You can use this metric to
+track the frequency of package reloads and set up alerts if reloads happen more often than expected.
+
+
+Dynamic package reload applies to Daml package deployments only. Other configuration changes such as adding
+parties, templates, or interfaces to filters still require a restart. See [Configure](/sdks-tools/development-tools/pqs/configure).
+
+
+
+
+
-### Exit codes
+#### Exit codes
PQS terminates with the following exit codes:
- `0`: Normal termination
+
- `1`: Termination due to unrecoverable error or all retry attempts for recoverable errors have been exhausted
### Ledger streaming & recovery
-On (re-)start, PQS determines last saved checkpoint and continues incremental processing from that point onward. PQS is able to start and finish at prescribed ledger offsets, specified via args.
+On (re-)start, PQS determines last saved checkpoint and continues incremental processing from that point onward. PQS
+is able to start and finish at prescribed ledger offsets, specified via args.
-In many scenarios `--pipeline-ledger-start Oldest --pipeline-ledger-stop Never` is the most appropriate configuration, for both initial population of all available history, and also catering for resumption/recovery processing.
+In many scenarios `--pipeline-ledger-start Oldest --pipeline-ledger-stop Never` is the most appropriate
+configuration, for both initial population of all available history, and also catering for resumption/recovery
+processing.
Start offset meanings:
-| Value | Meaning |
-|------------|---------------------------------------------------------------------------------------------------------|
-| `Genesis` | Commence from the first offset of the ledger, failing if not available. |
-| `Oldest` | Resume processing, or start from the oldest available offset of the ledger (if the datastore is empty). |
-| `Latest` | Resume processing, or start from the latest available offset of the ledger (if the datastore is empty). |
-| `` | Offset from which to start processing, terminating if it does not match the state of the datastore. |
+| Value | Meaning |
+| --- | --- |
+| `Genesis` | Commence from the first offset of the ledger, failing if not available. |
+| `Oldest` | Resume processing, or start from the oldest available offset of the ledger (if the datastore is empty). |
+| `Latest` | Resume processing, or start from the latest available offset of the ledger (if the datastore is empty). |
+| `\` | Offset from which to start processing, terminating if it does not match the state of the datastore. |
Stop offset meanings:
-| Value | Meaning |
-|------------|-----------------------------------------------------------------------------------|
-| `Latest` | Process until reaching the latest available offset of the ledger, then terminate. |
-| `Never` | Keep processing and never terminate. |
-| `` | Process until reaching this offset, then terminate. |
-
-
-
-If the ledger has been pruned beyond the offset specified in `--pipeline-ledger-start`, PQS fails to start. For more details see `pqs-history-slicing`.
-
-
+| Value | Meaning |
+| --- | --- |
+| `Latest` | Process until reaching the latest available offset of the ledger, then terminate. |
+| `Never` | Keep processing and never terminate. |
+| `\` | Process until reaching this offset, then terminate. |
-[^1]: [https://grpc.io/docs/guides/status-codes/](https://grpc.io/docs/guides/status-codes/)
+
+If the ledger has been pruned beyond the offset specified in `--pipeline-ledger-start`, PQS fails to start.
+For more details see [History slicing](/sdks-tools/development-tools/pqs/operate#history-slicing).
+
-[^2]: [https://github.com/pgjdbc/pgjdbc/blob/master/pgjdbc/src/main/java/org/postgresql/util/PSQLState.java](https://github.com/pgjdbc/pgjdbc/blob/master/pgjdbc/src/main/java/org/postgresql/util/PSQLState.java)
+[#] https://grpc.io/docs/guides/status-codes/
+[#] https://github.com/pgjdbc/pgjdbc/blob/master/pgjdbc/src/main/java/org/postgresql/util/PSQLState.java
{/* COPIED_END */}
## Secure
-{/* COPIED_START source="docs-website:docs/replicated/pqs/3.4/component-howtos/pqs/secure.rst" hash="6ca16d7c" */}
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/secure.rst" hash="3c9bf3c2" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
-PQS application is a client to backend services (ledger and database) as such it needs to respect security settings mandated by those services - TLS and authentication:
+PQS application is a client to backend services (ledger and database) as such it needs to respect security settings
+mandated by those services - TLS and authentication:
### TLS
-Your server-side components (Canton and PostgreSQL) may require TLS to be used. Please refer to their documentation for instructions:
+Your server-side components (Canton and PostgreSQL) may require TLS to be used. Please refer to their documentation
+for instructions:
-- for PostgreSQL see [https://www.postgresql.org/docs/current/ssl-tcp.html](https://www.postgresql.org/docs/current/ssl-tcp.html)
-- for Canton see `tls-configuration`
+- for PostgreSQL see https://www.postgresql.org/docs/current/ssl-tcp.html
+
+- for Canton see [TLS API Configuration](/global-synchronizer/reference/security-configuration#tls-api-configuration)
Once configured, use appropriate values for dedicated parameters:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-tls-cert /path/to/ledger.crt \
--source-ledger-tls-key /path/to/ledger.pem \
@@ -1202,9 +1515,10 @@ $ ./scribe.jar pipeline ledger postgres-document \
### Ledger authentication
-To run PQS with authentication you need to turn it on via `--source-ledger-auth OAuth`. PQS uses OAuth 2.0 Client Credentials flow[^1].
+To run PQS with authentication you need to turn it on via `--source-ledger-auth OAuth`. PQS uses OAuth 2.0 Client
+Credentials flow [#]_.
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-auth OAuth \
--pipeline-oauth-clientid my_client_id \
@@ -1215,7 +1529,7 @@ $ ./scribe.jar pipeline ledger postgres-document \
If your issuer is OIDC compliant, you can specify the issuer instead of the token URL.
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-auth OAuth \
--pipeline-oauth-clientid my_client_id \
@@ -1224,15 +1538,19 @@ $ ./scribe.jar pipeline ledger postgres-document \
--pipeline-oauth-issuer https://my-auth-server
```
-PQS uses the supplied client credentials (`clientid` and `clientsecret`) to access the token endpoint (`endpoint`) of the OAuth service of your choice. Optional `cafile` parameter is a path to the Certification Authority certificate used to access the token endpoint. If `cafile` is not set, the Java TrustStore is used.
+PQS uses the supplied client credentials (`clientid` and `clientsecret`) to access the token endpoint
+(`endpoint`) of the OAuth service of your choice. Optional `cafile` parameter is a path to the Certification
+Authority certificate used to access the token endpoint. If `cafile` is not set, the Java TrustStore is used.
-Please make sure you have configured your Participant Node to use authorization (see `ledger-api-jwt-configuration`) and an authorization server to accept your client credentials for `grant_type=client_credentials` and `scope=daml_ledger_api`.
+Please make sure you have configured your Participant Node to use authorization
+(see [Configure authorization service](/global-synchronizer/reference/security-configuration#configure-authorization-service)) and an authorization server to
+accept your client credentials for `grant_type=client_credentials` and `scope=daml_ledger_api`.
-### Audience-based token
+#### Audience-based token
For Audience-Based Tokens use the `--pipeline-oauth-parameters-audience` parameter:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-auth OAuth \
--pipeline-oauth-clientid my_client_id \
@@ -1243,11 +1561,11 @@ $ ./scribe.jar pipeline ledger postgres-document \
--pipeline-oauth-parameters-audience https://daml.com/jwt/aud/participant/my_participant_id
```
-### Scope-based token
+#### Scope-based token
For Scope-Based Tokens use the `--pipeline-oauth-scope` parameter:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-auth OAuth \
--pipeline-oauth-clientid my_client_id \
@@ -1259,28 +1577,37 @@ $ ./scribe.jar pipeline ledger postgres-document \
```
-The default value of the `--pipeline-oauth-scope` parameter is `daml_ledger_api`. Ledger API requires `daml_ledger_api` in the list of scopes unless custom target scope is configured.
+The default value of the `--pipeline-oauth-scope` parameter is `daml_ledger_api`. Ledger API requires
+`daml_ledger_api` in the list of scopes unless custom target scope is configured.
-### Custom Daml claims tokens
+#### Custom Daml claims tokens
-PQS authenticates as a user defined through the User Identity Management feature of Canton. Consequently, Custom Daml Claims Access Tokens are not supported. An audience-based or scope-based token must be used instead.
+PQS authenticates as a user defined through the User Identity Management feature of Canton. Consequently, Custom
+Daml Claims Access Tokens are not supported. An audience-based or scope-based token must be used instead.
-### Static access token
+#### Static access token
-Alternatively, you can configure PQS to use a static access token (meaning it is not refreshed) using the `--pipeline-oauth-accesstoken` parameter:
+Alternatively, you can configure PQS to use a static access token (meaning it is not refreshed) using the
+`--pipeline-oauth-accesstoken` parameter:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-auth OAuth \
--pipeline-oauth-accesstoken my_access_token
```
-### Ledger API users and Daml parties
+#### Ledger API users and Daml parties
-PQS connects to a Participant Node (via Ledger API) as a user defined through the User Identity Management feature of Canton. PQS gets its user identity by providing an OAuth token of that user. After authenticating, the Participant Node has the authorization information to know what Daml Party data the user is allowed to access. By default, PQS will subscribe to data for all parties available to PQS' authenticated user. However, this scope can be limited via the `--pipeline-filter-parties` filter parameter (see `pqs-party-filtering`).
+PQS connects to a Participant Node (via Ledger API) as a user defined through the User Identity Management feature of Canton.
+PQS gets its user identity by providing an OAuth token of that user. After authenticating, the Participant Node has the
+authorization information to know what Daml Party data the user is allowed to access. By default, PQS will subscribe
+to data for all parties available to PQS' authenticated user. However, this scope can be limited via the
+`--pipeline-filter-parties` filter parameter (see [Party filtering](/sdks-tools/development-tools/pqs/configure#party-filtering)).
-It is important to keep in mind that a PQS database instance might contain data for multiple Daml parties. To that extent, it is of paramount significance to ensure that queries are always scoped to the relevant parties, to avoid data leaks:
+It is important to keep in mind that a PQS database instance might contain data for multiple Daml parties. To that
+extent, it is of paramount significance to ensure that queries are always scoped to the relevant parties, to avoid
+data leaks:
```sql
-- partyA needs to be signatory on the contract
@@ -1299,17 +1626,25 @@ select a.* from active() a where stakeholders(a.*) @> '{partyC}';
select * from creates() where divulged_only and witnesses @> '{partyD}';
```
-### Token expiry
+#### Token expiry
-JWT tokens[^2] have an expiration time. PQS has a mechanism to automatically request a new access token from the Auth Server, before the old access token expires. To set when PQS should try to request a new access token, use `--pipeline-oauth-preemptexpiry` (default "PT1M" - one minute), meaning: request a new access token one minute before the current access token expires. This new access token is used for any future Ledger API calls.
+JWT tokens [#]_ have an expiration time. PQS has a mechanism to automatically request a new access token from the Auth
+Server, before the old access token expires. To set when PQS should try to request a new access token, use
+`--pipeline-oauth-preemptexpiry` (default "PT1M" - one minute), meaning: request a new access token one minute
+before the current access token expires. This new access token is used for any future Ledger API calls.
-However, for streaming calls such as GetUpdates the access token is part of the request that initiates the streaming. Canton versions prior to `2.9` terminate the stream with error `PERMISSION_DENIED` as soon as the old access token expires to prevent streaming forever based on the old access token. Versions `2.9+` fail with code `ABORTED` and description `ACCESS_TOKEN_EXPIRED` and PQS streams from the offset of the last successfully processed transaction.
+However, for streaming calls such as [GetUpdates](/reference/grpc-ledger-api-reference/com-daml-ledger-api-v2/updateservice/getupdates) the
+access token is part of the request that initiates the streaming. Canton versions prior to `2.9` terminate the
+stream with error `PERMISSION_DENIED` as soon as the old access token expires to prevent streaming forever based on
+the old access token. Versions `2.9+` fail with code `ABORTED` and description `ACCESS_TOKEN_EXPIRED` and PQS
+streams from the offset of the last successfully processed transaction.
-### Forward proxy
+#### Forward proxy
-If PQS runs in a network that requires a forward proxy to reach external OAuth endpoints (for example Azure AD or Okta), configure the proxy using CLI flags:
+If PQS runs in a network that requires a forward proxy to reach external OAuth endpoints
+(for example Azure AD or Okta), configure the proxy using CLI flags:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--source-ledger-auth OAuth \
--pipeline-oauth-clientid my_client_id \
@@ -1322,25 +1657,30 @@ $ ./scribe.jar pipeline ledger postgres-document \
Or via environment variables:
-``` text
+```text
SCRIBE_PIPELINE_OAUTH_PROXY_URL=http://proxy.corp.example.com:8080
SCRIBE_PIPELINE_OAUTH_PROXY_USER=proxyuser # optional
SCRIBE_PIPELINE_OAUTH_PROXY_PASSWORD=proxypass # optional
```
-The proxy is used for both OIDC discovery and token acquisition requests. It does not affect the gRPC connection to the ledger API or the PostgreSQL connection.
+The proxy is used for both OIDC discovery and token acquisition requests. It does not affect
+the gRPC connection to the ledger API or the PostgreSQL connection.
-If your infrastructure uses `HTTPS_PROXY` environment variables, set `SCRIBE_PIPELINE_OAUTH_PROXY_URL` to the same value.
+If your infrastructure uses `HTTPS_PROXY` environment variables, set `SCRIBE_PIPELINE_OAUTH_PROXY_URL`
+to the same value.
-If PQS runs inside an Istio service mesh, the Istio sidecar intercepts outbound TCP connections. For proxy connectivity to work, either create an Istio `ServiceEntry` for the proxy host, or exclude the proxy port from sidecar interception using the pod annotation `traffic.sidecar.istio.io/excludeOutboundPorts`.
+If PQS runs inside an Istio service mesh, the Istio sidecar intercepts outbound TCP connections.
+For proxy connectivity to work, either create an Istio `ServiceEntry` for the proxy host, or
+exclude the proxy port from sidecar interception using the pod annotation
+`traffic.sidecar.istio.io/excludeOutboundPorts`.
### PostgreSQL authentication
To authenticate to PostgreSQL, use dedicated parameters when launching the pipeline:
-``` text
+```text
$ ./scribe.jar pipeline ledger postgres-document \
--target-postgres-password "${YOUR_DB_PASSWORD}" \
--target-postgres-username "${YOUR_DB_USER}"
@@ -1348,56 +1688,96 @@ $ ./scribe.jar pipeline ledger postgres-document \
### Hardening recommendations
-**Use TLS**: Always use TLS to encrypt data being transmitted to/from the Participant Node and the PostgreSQL datastore. This is especially important when dealing with sensitive information. Ensure that only secure TLS versions are used (for example TLS `1.2+`) and that strong cipher suites are configured. Client authentication should be used to ensure that only trusted clients can connect to the Participant Node and PostgreSQL datastore, such that network level security is not overly relied upon.
+**Use TLS**: Always use TLS to encrypt data being transmitted to/from the Participant Node and the PostgreSQL
+datastore. This is especially important when dealing with sensitive information. Ensure that only secure TLS versions
+are used (for example TLS `1.2+`) and that strong cipher suites are configured. Client authentication should be used to
+ensure that only trusted clients can connect to the Participant Node and PostgreSQL datastore, such that network
+level security is not overly relied upon.
-**Logging**: Ensure that logging is configured (see `pqs-logging`) to avoid logging sensitive information. This includes transaction details and metadata (eg. size) that is revealed in `TRACE` and `DEBUG` levels. These log levels should be used with caution, and only in a controlled environment. In production, we recommend using `INFO` or `WARN` levels.
+**Logging**: Ensure that logging is configured (see [Logging](#logging)) to avoid logging sensitive information. This
+includes transaction details and metadata (eg. size) that is revealed in `TRACE` and `DEBUG` levels. These log
+levels should be used with caution, and only in a controlled environment. In production, we recommend using `INFO`
+or `WARN` levels.
-**Ledger Authorization**: Follow the principle of least privilege when granting access to the Participant Node Ledger API User that PQS uses:
+**Ledger Authorization**: Follow the principle of least privilege when granting access to the Participant Node
+Ledger API User that PQS uses:
- Only `canReadAs` authorization to only Party's that it requires; OR
-- Only `readAsAnyParty` authorization if PQS is used as a participant-wide service and needs access to all Party data.
+
+- Only `readAsAnyParty` authorization if PQS is used as a participant-wide service and needs access to all Party
+ data.
+
- No `canActAs` authorization (to submit commands). PQS has no capability to submit commands to the Canton Ledger API
+
- No `admin` access to the Canton Ledger API.
-**Database Access**: The datastore contains all ledger information obtained from the Participant Node. Ensure that the database users are tightly controlled, and set to the minimum required privileges:
+**Database Access**: The datastore contains all ledger information obtained from the Participant Node. Ensure that
+the database users are tightly controlled, and set to the minimum required privileges:
+
+- Operational user: SQL Insert/Update/Delete/Copy - so PQS can maintain the datastore contents. No DDL rights should
+ be in place - PQS does not need to change the database schema.
+
+- Others user: No write access should be granted to any other user. Excessive reading my other clients (leading to
+ overload) should be avoided, to ensure that PQS has sufficient resources to operate.
-- Operational user: SQL Insert/Update/Delete/Copy - so PQS can maintain the datastore contents. No DDL rights should be in place - PQS does not need to change the database schema.
-- Others user: No write access should be granted to any other user. Excessive reading my other clients (leading to overload) should be avoided, to ensure that PQS has sufficient resources to operate.
-- Admin user: PQS will need to be able to apply schema changes to the database when deploying a new version containing database changes. This should be a separate user with the minimum required privileges to perform these operations. Also, redaction operations performed by an administrator will require Select/Update rights to the database.
+- Admin user: PQS will need to be able to apply schema changes to the database when deploying a new version
+ containing database changes. This should be a separate user with the minimum required privileges to perform these
+ operations. Also, redaction operations performed by an administrator will require Select/Update rights to the
+ database.
**Network Security**: Ensure that the network security is configured to restrict access to only essential connections:
- Using firewalls to restrict access to PostgreSQL database, Participant Node and auth server.
-- Using firewalls to restrict access to PQS. Even though it does not listen for any client connections, there are listing TCP ports for health and diagnostic purposes. By default, health and diagnostic ports are only accessible from `localhost`. Before changing this configuration, ensure that all hosts granted network level access are necessary & trusted: to mitigate the risk of exploit or exposing sensitive information.
+
+- Using firewalls to restrict access to PQS. Even though it does not listen for any client connections, there are
+ listing TCP ports for health and diagnostic purposes. By default, health and diagnostic ports are only accessible
+ from `localhost`. Before changing this configuration, ensure that all hosts granted network level access are
+ necessary & trusted: to mitigate the risk of exploit or exposing sensitive information.
**Runtime Environment**: Ensure that the runtime environment is secure:
- Keeping the operating system and all software up to date with security patches.
+
- Using a firewall to restrict access to the PQS process (and associated PostgreSQL datastore).
+
- Monitoring logs for any suspicious activity, as well as errors and warnings.
-- Validate that the runtime enforces least privilege principles, and contains only intended tools. We recommend using a minimal Java runtime environment (JRE) to reduce the attack surface:
+
+- Validate that the runtime enforces least privilege principles, and contains only intended tools. We recommend using
+ a minimal Java runtime environment (JRE) to reduce the attack surface:
+
- Use a minimal JRE (not JDK), for example Amazon Corretto or Azul Zulu, to reduce the attack surface.
- - Consider allowing only the `jdk.attach`[^3] module, in your chosen JRE. This enables the running process to produce more accurate stack traces when diagnostics are extracted.
+
+ - Consider allowing only the `jdk.attach` [#]_ module, in your chosen JRE. This enables the running process to
+ produce more accurate stack traces when diagnostics are extracted.
+
- Use a security manager to restrict the infrastructure permissions of the PQS process, if possible.
+
- Use a containerized environment (for example Docker) to isolate the PQS process from the host system and other processes.
-**Observability**: Ensure that the environment is monitored for logs, metrics and alerts of events of interest (see `pqs-observe`).
+**Observability**: Ensure that the environment is monitored for logs, metrics and alerts of events of interest (see
+[Observe](#observe)).
- PQS log is monitored for errors and warnings, to ensure these do not go unnoticed.
+
- PQS runtime is monitored for disk, memory and other JRE concerns such as heap, garbage collection cycle rates.
-- PQS health endpoint is polled regularly to verify availability.
-- PostgreSQL and Participant Node servers are similarly monitored.
-**Database Backups**: Ensure that the database backups are encrypted and stored securely. This is especially important when dealing with sensitive information. The database should be backed up regularly, and the backup process should be tested to ensure that it works as expected.
+- PQS health endpoint is polled regularly to verify availability.
-**Data Retention**: Ensure that the data retention policy is in place and enforced. This includes regular purging of old data (eg. PQS `pqs-pruning`), and ensuring that sensitive information is redacted or deleted as required by your organization's policies and regulations.
+- PostgreSQL and Participant Node servers are similarly monitored.
-**Software Updates**: Ensure that the PQS software is kept up to date with the latest security patches and updates. This includes both the PQS software itself and any dependencies it may have.
+**Database Backups**: Ensure that the database backups are encrypted and stored securely. This is especially
+important when dealing with sensitive information. The database should be backed up regularly, and the backup process
+should be tested to ensure that it works as expected.
-[^1]: [https://datatracker.ietf.org/doc/html/rfc6749#section-4.4](https://datatracker.ietf.org/doc/html/rfc6749#section-4.4)
+**Data Retention**: Ensure that the data retention policy is in place and enforced. This includes regular purging of
+old data (eg. PQS [Pruning](/sdks-tools/development-tools/pqs/operate#pruning)), and ensuring that sensitive information is redacted or deleted as required by
+your organization's policies and regulations.
-[^2]: [https://datatracker.ietf.org/doc/html/rfc6749#section-1.5](https://datatracker.ietf.org/doc/html/rfc6749#section-1.5)
+**Software Updates**: Ensure that the PQS software is kept up to date with the latest security patches and updates.
+This includes both the PQS software itself and any dependencies it may have.
-[^3]: [https://docs.oracle.com/en/java/javase/17/docs/api/jdk.attach/module-summary.html](https://docs.oracle.com/en/java/javase/17/docs/api/jdk.attach/module-summary.html)
+[#] https://datatracker.ietf.org/doc/html/rfc6749#section-4.4
+[#] https://datatracker.ietf.org/doc/html/rfc6749#section-1.5
+[#] https://docs.oracle.com/en/java/javase/17/docs/api/jdk.attach/module-summary.html
{/* COPIED_END */}
diff --git a/docs-main/sdks-tools/development-tools/pqs/troubleshoot.mdx b/docs-main/sdks-tools/development-tools/pqs/troubleshoot.mdx
new file mode 100644
index 000000000..0d33d76f3
--- /dev/null
+++ b/docs-main/sdks-tools/development-tools/pqs/troubleshoot.mdx
@@ -0,0 +1,742 @@
+---
+title: "Troubleshoot"
+description: "Diagnose common PQS issues using quick facts, log and metrics analysis, and step-by-step runbooks."
+---
+
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/troubleshoot/index.rst" hash="5211dbbd" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
+
+When things go wrong, it is important to be able to diagnose the problem quickly and effectively. This section aims
+to get an operator up to speed with the most common troubleshooting techniques and tools available in PQS. You might
+want to refer to it for ideas when devising your own troubleshooting procedures.
+
+PQS application (`pipeline` process) quick facts:
+
+- exports ledger events into queryable data store
+
+- does not send ledger commands
+
+- is stateless
+
+- is restart [friendly](/sdks-tools/development-tools/pqs/operate#retries) (fast restarts in absence of migration, Daml model changes, etc)
+
+- is tolerant to unavailable dependencies (through retry loop)
+
+- uses only 1 Ledger API stream connection (flat transaction or transaction tree) after
+ [initialisation](#why-might-it-take-a-long-time-before-pqs-starts-processing-streams)
+
+- uses a pool of connections to Postgres (16 by default)
+
+- can be secured with TLS on both connections
+
+- uses OpenTelemetry [Agent](/sdks-tools/development-tools/pqs/operate#observe) for its observability signals exports
+
+- can export [diagnostics](/sdks-tools/development-tools/pqs/operate#diagnostics) archive (with metrics and thread dumps over time)
+
+Look into exit codes of PQS process or Docker/Kubernetes container orchestrator:
+
+- `137` indicates the process was killed by external forces with
+ `SIGKILL` (`-9`), see [also](#jvm-metrics)
+
+- non-zero exit might indicate invalid starting conditions which are treated as non-recoverable errors. Causes might
+ include:
+
+ - misspelled startup [parameter](#pqs-does-not-accept-the-passed-in-settings) names or values
+
+Look into logs for activity indicators:
+
+- ledger [keep-alives](#there-is-a-suspicion-that-pqs-is-stalled) are present
+
+- watermark advances in the presence of expected ledger traffic, see also [here](#don-t-see-expected-templates-in-pqs) and [here](#no-data-for-a-recently-onboarded-party)
+
+- retry loop indicates [recoverable errors](/sdks-tools/development-tools/pqs/operate#retries) (both upstream and downstream), examine message for
+ indication of underlying cause
+- in case of non-recoverable errors, keep in mind that the last visible stacktrace does not necessarily represent the
+ true root cause - explore events that preceded it by requesting a bigger slice of logs before the termination
+
+Look into [metrics](#which-metrics-are-available-how-to-integrate-them) for detailed breakdown of PQS
+internals:
+
+- correlate [transactions](#throughput-transactions-and-events)
+ (`pipeline_events_total{type="transaction"}`) and [watermark](#throughput-watermark-history)
+ (`watermark_ix`) throughput metrics to identify if any slowdowns are present in the PQS pipeline
+
+- get an idea of PQS pipeline introduced latency - see
+ [here](#latency-total-transaction-handling-latency)
+
+- get an idea of contract churn (which correlates with write activity of PQS) by template - see
+ [here](#contracts-churn)
+
+Look into database to get familiar with Daml model footprint:
+
+- get an idea of data volumes in terms of Daml structure - see [here](#all-non-empty-tables-rows-count)
+ and [here](#total-sizes-of-tables)
+
+Look into database statistics for resource utilisation
+
+- get an idea of I/O split - [disk vs index](#statistics-on-disk-vs-index-i-o), cache sizing
+ ([tables](#table-data-in-cache-per-table)
+ and [indexes](#index-data-in-cache-per-table))
+
+- probe for heavy queries ([current](#currently-executing-queries) and
+ [over time](#top-10-queries-by-run-time))
+
+- inspect if `bgwriter` [flush](#bgwriter-frequency) triggers too frequently
+
+Try correlating representative metrics between PQS & Canton (if available).
+
+To escalate issues to Digital Asset's support team, please provide forensics by collecting
+[diagnostics](/sdks-tools/development-tools/pqs/operate#diagnostics) dump in proximity of the incident time and attach the resulting archive to the
+support ticket.
+
+## Runbooks
+
+{/* COPIED_END */}
+
+{/* COPIED_START source="docs-website:docs/replicated/pqs/3.5/component-howtos/pqs/troubleshoot/runbooks.rst" hash="393232e9" */}
+
+{/* Copyright (c) 2025, Digital Asset (Switzerland) GmbH and/or its affiliates. All rights reserved. */}
+
+Runbooks are recipes that provide instructions for handling common issues or tasks. They are designed to be easy to
+follow and should include all necessary steps to resolve an issue.
+
+### PQS does not accept the passed in settings
+
+Check if the argument in question is in the supported list of arguments by re-running the base command with:
+
+- `--help` / `-h` option for command-line arguments
+
+- `--help-verbose` / `-H` option for command-line arguments, environment variables and Java system properties
+
+Check the settings for spelling and capitalisation.
+
+Check if the argument supplied is applied and quoted in the configuration banner in the logs.
+
+For example,
+
+```text
+$ ./scribe.jar pipeline ledger postgres-document --source-ledger-host 10.0.0.0
+```
+
+should result in the following banner being displayed in the logs:
+
+```text
+Applied configuration:
+...
+source {
+ ledger {
+ ...
+ host=10.0.0.0
+ ...
+ }
+}
+```
+
+### Adjusting PQS settings dynamically (at runtime)
+
+PQS does not allow one to adjust any of its settings on the fly. Configuration settings along with the dynamically
+resolved party filters and interface filters are fixed at the start of pipeline execution. Any change
+requires a restart, which should be fairly cheap under normal circumstances. You might also want to re-visit the
+configuration [warning](/sdks-tools/development-tools/pqs/configure).
+
+
+Daml packages are an exception: when PQS detects a new package when processing a Daml payload received from the Ledger API, it reloads itself and fetches the missing packages (DARs) from the Participant Node. See [Dynamic Daml package reload](/sdks-tools/development-tools/pqs/operate#dynamic-daml-package-reload).
+
+
+### There is a suspicion that PQS is stalled
+
+Check for ledger keep-alive output in the logs (frequency configured via `--source-ledger-keepalive` (default 40s)).
+
+You should at least see this line repeated every 40 seconds, even if there is no new activity on the ledger:
+
+```text
+14:42:28.281 I [zio-fiber-600078685] com.digitalasset.zio.daml.Channel:37 Keep-alive (get ledger version) successful
+```
+
+If you expect data to flow, then check your filter configuration so that it’s not too restrictive (see
+[Don’t see expected templates in PQS](#don-t-see-expected-templates-in-pqs)).
+
+If all else fails, get your hands on a thread dump (for example, through [diagnostics](/sdks-tools/development-tools/pqs/operate#diagnostics)) and
+analyse it for any deadlocks.
+
+### Why might it take a long time before PQS starts processing streams?
+
+The following phases happen during PQS startup:
+
+- fetching of all DARs from the ledger
+
+- parsing DARs locally to extract type information (cached locally for subsequent restarts)
+
+- converting type information into codecs
+
+- initializing DB tables and partitions corresponding to templates/interfaces and exercises
+
+- processing ACS if applicable
+
+- pipeline with ongoing processing starts now
+
+Actual start-up time may be affected by multiple reasons, including, but
+not limited to:
+
+- slow network
+
+- excessive number of DARs/packages on ledger (not yet cached by PQS)
+
+- excessive number of new templates (likely caused by Daml upgrade/migration procedures)
+
+- the size of ACS (if applicable)
+
+- processing from early offsets (historical data) on a very lengthy ledger (will influence until PQS catches up to
+ the head for ongoing streaming)
+
+### Don’t see expected templates in PQS
+
+Check that your filter’s configuration is not too restrictive. Sanity check the overall count in `INFO`-level logs
+
+```text
+Applied configuration:
+...
+pipeline {
+ filter {
+ contracts="*"
+ ...
+ }
+}
+...
+I [zio-fiber-330812857] com.digitalasset.zio.daml.ledgerapi.logFilterContents:12 Contract filter inclusive of 1 templates and 1 interfaces
+```
+
+Running logging at `DEBUG` [level](/sdks-tools/development-tools/pqs/operate#logging) (`--logger-level=Debug`) will provide more detailed
+information about specific templates included in the synchronization pipeline:
+
+```text
+14:52:40.201 D [zio-fiber-655477872] com.digitalasset.zio.daml.ledgerapi.logFilterContents:13 Including template 5fe0c850054845d95a855650cba76d9d999c3e985ae549bcca01253b17b08b2c:PingPong:Ping
+14:52:40.201 D [zio-fiber-655477872] com.digitalasset.zio.daml.ledgerapi.logFilterContents:13 Including template 5fe0c850054845d95a855650cba76d9d999c3e985ae549bcca01253b17b08b2c:PingPong:PingWithCK
+14:52:40.201 D [zio-fiber-655477872] com.digitalasset.zio.daml.ledgerapi.logFilterContents:13 Including template 5fe0c850054845d95a855650cba76d9d999c3e985ae549bcca01253b17b08b2c:PingPong:Pong
+```
+
+### Is it safe to change PQS `--pipeline-datasource` against the data store with existing data?
+
+You might want to re-visit the configuration
+[warning](/sdks-tools/development-tools/pqs/configure).
+
+### Debug output is too noisy
+
+To prevent excessive output from Netty when the logging level is set to `DEBUG`, use the following arguments:
+
+```text
+$ ./scribe.jar pipeline \
+ --logger-level=Debug \
+ --logger-mappings-io.netty=Info \
+ --logger-mappings-io.grpc.netty=Info
+```
+
+### No data for a recently onboarded party
+
+PQS is not notified of new parties, so it needs a restart to acknowledge the new party set.
+
+Note that the best approach when onboarding new parties is to:
+
+- stop PQS from processing the data
+
+- onboard the new party
+
+- start PQS processing where it left off
+
+Otherwise, there is a chance of corrupting PQS data for the newly onboarded party if party onboarding happens during
+the active PQS pipeline.
+
+If a new party was onboarded without stopping the PQS pipeline, it is best to either:
+
+- purge the PQS database and re-ingest the data either from `Genesis` or from `Latest`
+ (see [Ledger streaming & recovery](/sdks-tools/development-tools/pqs/operate#ledger-streaming-recovery)), or
+
+- perform a reset from a particular offset (before the offset of the first event with the new party involved) by
+ following the [instructions](/sdks-tools/development-tools/pqs/operate#resetting)
+
+The list of parties in the current pipeline session is output in the logs on startup:
+
+```text
+15:01:40.872 I [zio-fiber-950510203] com.digitalasset.zio.daml.ledgerapi.PartiesService:61 2 known parties retrieved
+15:01:40.876 I [zio-fiber-950510203] com.digitalasset.scribe.pipeline.pipeline.Impl:39 Starting pipeline on behalf of 'Alice::12209adab9c5e9d672d7b4515e26f4cd296cc2ec99bdb9787be7e804c49ca52f2686,Bob::12209adab9c5e9d672d7b4515e26f4cd296cc2ec99bdb9787be7e804c49ca52f2686'
+```
+
+### PQS complains it cannot start due to various offset mismatches
+
+PQS may fail to start due to a non-reconcilable gap in the events history (potentially caused by ledger pruning or
+other factors). Please, refer to these 2 sections for additional insights:
+
+- [History slicing](/sdks-tools/development-tools/pqs/operate#history-slicing)
+
+- [Ledger streaming & recovery](/sdks-tools/development-tools/pqs/operate#ledger-streaming-recovery)
+
+In most likelihood, under normal conditions, PQS should be launched with the following arguments, unless there is a
+strong reason to modify them:
+
+```text
+$ ./scribe.jar pipeline --pipeline-ledger-start=Oldest --pipeline-ledger-stop=Never
+```
+
+Please, refer to the logs for the steps of determination, which offset is being selected to start the pipeline
+
+```text
+15:28:12.778 I [zio-fiber-1984883840] com.digitalasset.zio.daml.ledgerapi.StateService:79 Retrieved ledger end offset: 00000000000000000f
+15:28:12.809 I [zio-fiber-1984883840] com.digitalasset.scribe.pipeline.pipeline.Impl:91 Last known checkpoint is at offset '00000000000000000f' and index '9'
+15:28:12.811 I [zio-fiber-1984883840] com.digitalasset.scribe.pipeline.pipeline.Impl:100 Continuing from offset '00000000000000000f' and index '9' until offset 'INFINITY'
+```
+
+### Is it safe to restart PQS? Can data get corrupted?
+
+PQS was designed with failure friendliness - it does not require graceful shutdown or draining of active tasks. It is
+absolutely fine if the JVM process gets killed. On a restart, PQS will perform a clean-up procedure of data beyond
+the current watermark and then re-subscribe and continue processing from the watermark’s offset onwards (see
+[Recover](/sdks-tools/development-tools/pqs/operate#recover)).
+
+### What happens if multiple PQS instances are launched against the same data store?
+
+In case multiple PQS instances are launched against the same data store, no data corruption happens. However, they
+will be competing among themselves to become the exclusive writer, so it might affect the throughput of ledger stream
+consumption. It is **unadvisable** to do so (see also [High availability](/sdks-tools/development-tools/pqs/operate#high-availability)).
+
+### Which metrics are available? How to integrate them?
+
+PQS Docker images are published with the preconfigured Prometheus endpoint (see [Observe](/sdks-tools/development-tools/pqs/operate#observe)). By default,
+it listens on `0.0.0.0:9090`, so it is expected to be hooked into existing monitoring infrastructure.
+
+To that extent, PQS releases are accompanied by [dashboards](/sdks-tools/development-tools/pqs/operate#grafana-dashboard) that can be imported into a
+Grafana instance for convenient visualisation of application health.
+
+It is highly advised to implement an observability platform in your environment. While logs help troubleshoot
+correctness issues, metrics are much more suitable for performance related troubleshooting.
+
+### PQS processing throughput seems low
+
+PQS had been benchmarked against high throughput scenarios of up to 100K+ events/sec. PQS processing pipeline is just an
+intermediary between two data systems - ledger (Canton) and relational database (Postgres), so most likely to
+troubleshoot such a cause, one would need to dig into one of these endpoints.
+
+Suggested areas of attention (see also [Optimize](/sdks-tools/development-tools/pqs/optimize)):
+
+- ledger
+
+ - host/OS-level metrics - CPU, I/O, RAM, etc (look for resource saturation)
+ - check with Canton for relevant metrics
+
+- Postgres
+
+ - check minimum resources requirements for Postgres
+
+ - check host/OS-level metrics - CPU, I/O, RAM, etc (look for resource saturation)
+
+ - configuration settings and non-default overrides
+
+ - metrics according to database (`pg_stat_statements`, `pg_stat_activity`)
+
+- PQS
+
+ - check minimum resources requirements for PQS to avoid unexpected imbalance
+
+ - check host/OS-level metrics - CPU, I/O, RAM, etc (look for resource saturation)
+
+ - ensure these settings are not set too low (no less than defaults)
+
+ - `--source-ledger-buffersize`
+
+ - `--target-postgres-buffersize`
+
+ - `--target-postgres-maxconnections` (because parallel processing which affects both throughput and latency
+ is tied to this configuration)
+
+ - check vital metrics look good ([PQS metrics dashboard and how to read it](#pqs-metrics-dashboard-and-how-to-read-it))
+
+### Dissecting the logs
+
+PQS emits a healthy volume of relevant information while running into `stdout` stream. `INFO` level describes
+application-level events such as:
+
+- ledger keep-alives
+
+- lifecycle information
+
+- authentication events
+
+- starting conditions - offsets, current watermark, etc
+
+- ingress of ledger events
+
+- conversion of payloads
+
+- watermark advancement
+
+`DEBUG` and `TRACE` levels add supplementary troubleshooting information. Caution needs to be exercised since
+increasing the log level might affect performance negatively along with exposing sensitive data (contract’s contents,
+for instance).
+
+### PQS metrics dashboard and how to read it
+
+PQS [Dashboard](/sdks-tools/development-tools/pqs/operate#grafana-dashboard) is designed to read from top to bottom. It provides information from
+general to more specific, so it is good practice to scan through the dashboard as it goes and spot anomalies along
+the flow.
+
+#### `Contracts > Churn`
+
+Per-template activity (creates/archives) on the ledger. This chart may be useful to gauge relative throughputs in
+business terms.
+
+
+
+#### `Contracts > Active`
+
+Per-template active contracts count. This chart may be useful to gauge composition of ACS
+
+
+
+#### `Throughput > Throughputs`
+
+Current throughputs in terms of watermark advancement and stored events
+
+
+
+#### `Throughput > Ingested counts`
+
+Total counts (as measured in transactions and events dimensions) since latest PQS start
+
+
+
+#### `Throughput > Transaction lag`
+
+Tracks lag from ledger (delta between command completion determined by transaction’s `effective_at` attribute and
+ingestion by PQS pipeline as determined by wall clock). This is latency introduced by upstream processes outside of
+PQS control (Ledger API, network, etc). This chart indicates, for example, that 100 ms of latency has already been
+contributed to the overall end-to-end processing latency:
+
+
+
+#### `Throughput > Watermark history`
+
+Time series of watermark progression throughput. This manifests the rate of ledger transactions becoming available
+for querying with PQS Read API functions. The typical shape of the chart is shown below. For uniform traffic it
+should represent smooth curves.
+
+
+
+Anomalies might include torn shapes and zigzag patterns with inactivities followed by spikes.
+
+#### `Throughput > Transactions and events`
+
+The shape of ingested traffic in terms of transactions and events dimensions. Provides an idea of the coarseness of
+transaction sizes.
+
+
+
+#### `Throughput > Events breakdown`
+
+Provides breakdown of event types inside transactions (contents differs depending on pipeline source - flat
+transaction vs transaction tree)
+
+
+
+#### `Throughput > Waitpoints - ACS / streaming`
+
+*ACS = only during seeding from the ActiveContractSet Ledger API service*
+
+*streaming = normal processing pipeline*
+
+The internal pipeline is composed of distinct stages separated by queues (aka wait points). This chart indicates
+throughputs of items passing through them. Note that items might be distributed and consolidated at various stages,
+therefore relative throughputs can differ between the stages even for streamlined flow.
+
+
+
+Anomalies here might include change in relative throughputs indicating non-uniformity of neighbouring transaction
+sizes at certain points. This might indicate noisy neighbours potentially causing latency spikes for other transactions.
+
+#### `Queue sizes`
+
+Time-series of histograms (vertical slices) that represent queue size of named wait points. Bottom all-green line
+indicates queue was empty all the time and this represents healthy situation without queueing or back-pressuring:
+
+
+
+Below is a case where queueing was present due to downstream backpressure. This should be a matter of interest and
+suggests further investigation. Queue sizes are arranged cascadingly and the point at which queue size becomes empty
+points to a bottleneck, because pushback is propagated upstream.
+
+
+
+#### `Latency`
+
+Time taken to transfer a unit of work between wait points with different levels of granularity. Provides a
+percentile-based view as well historical histogram heatmap.
+
+In this chart we can observe from the left panel that the majority of operations take less than 10 ms with some
+outliers taking up to 30 ms. On the right chart we see more detailed insights, namely, there are two dominating
+operations with average latencies 1 ms and 5 ms each.
+
+
+
+Anomalies might include huge differences between p50 and p95 percentiles. As well as non-uniform spread of latencies
+in the right part.
+
+Slowly increasing latencies across a lengthy time slice should cause concerns of service degradation.
+
+#### `Latency > Total Transaction Handling Latency`
+
+Time taken by the entire PQS pipeline between receipt from Ledger API to being committed to Postgres.
+
+
+
+#### `JVM Metrics`
+
+Provides a series of metrics that are common across any JVM application in terms of memory management, CPU
+utilisation and garbage collection activity. The most crucial signals to monitor and interpret are:
+
+
+
+Anomalies might include `used` approaching `committed` and never decreasing. At the same time garbage collection
+frequency and time taken are increasing. These are the symptoms that out of memory conditions are imminent. If these
+parameters look healthy but PQS still exists with `137` exit code, then most likely a supervisor (Docker,
+Kubernetes) is terminating PQS forcefully - investigate potential configuration imbalance. Make sure that JVM
+memory-related [settings](/sdks-tools/development-tools/pqs/optimize#java-virtual-machine-configuration) are applied.
+
+### SQL queries useful for troubleshooting
+
+It is highly recommended that Postgres metrics are captured through appropriate tools (like
+[postgres-exporter](https://github.com/prometheus-community/postgres_exporter)) and shipped into metrics
+storage (like [Prometheus](https://prometheus.io/)) for historical trends identification and analysis. The
+following visualisations useful for analysis would be possible for creation:
+
+
+
+
+
+In the absence of integration into metrics storage, one can run the following queries when necessary for
+point-in-time view.
+
+#### Statistics on disk vs index I/O
+
+```sql
+select stat.relname as relname,
+ seq_scan,
+ seq_tup_read,
+ idx_scan,
+ idx_tup_fetch,
+ heap_blks_read,
+ heap_blks_hit,
+ round((100 * heap_blks_hit::float / coalesce(nullif(heap_blks_hit + heap_blks_read, 0), 1))::numeric, 2) as heap_blks_ratio,
+ idx_blks_read,
+ idx_blks_hit,
+ round((100 * idx_blks_hit::float / coalesce(nullif(idx_blks_hit + idx_blks_read, 0), 1))::numeric, 2) as idx_blks_ratio
+from pg_stat_user_tables stat
+ right join pg_statio_user_tables statio on stat.relid = statio.relid;
+```
+
+Look for rows whose heap or index ratio diverge from 100 but the number of reads or scans is significant.
+
+#### Currently executing queries
+
+```sql
+select pid,
+ datname,
+ usename,
+ application_name,
+ client_hostname,
+ client_port,
+ backend_start,
+ query_start,
+ (now() - query_start) as exec_time_so_far,
+ query,
+ state
+from pg_stat_activity
+where state = 'active';
+```
+
+#### Top 10 queries by run time
+
+This query requires `pg_stat_statements`
+[extension](https://www.postgresql.org/docs/current/pgstatstatements.html) installed into PostgreSQL.
+
+```sql
+select total_exec_time,
+ calls,
+ query,
+ queryid,
+ toplevel
+from pg_stat_statements
+where not (
+ query ilike 'create %' or
+ query ilike 'alter %' or
+ query ilike 'drop %' or
+ query ilike 'grant %' or
+ query ilike 'revoke %' or
+ query ilike 'set %' or
+ query ilike 'do %' or
+ query like '%pg_stat%' or
+ query like '%pg_database%' or
+ query like '%pg_replication%' or
+ query like '%information_schema%' or
+ query like '%prune_contracts%' or
+ query like 'with deleted_transactions%' or
+ query like 'with deleted_contracts%' or
+ query = 'SELECT version()' or
+ query = 'SELECT $1' or
+ query = 'BEGIN' or
+ query = 'COMMIT' or
+ query = 'ROLLBACK'
+ )
+order by total_exec_time desc
+limit 10;
+```
+
+#### All non-empty tables rows count
+
+```sql
+select schemaname, relname, n_live_tup as rows_in_table
+from pg_stat_user_tables
+where n_live_tup > 0
+order by rows_in_table desc, relname;
+```
+
+#### Table data in cache per table
+
+```sql
+select *
+from (select relname,
+ heap_blks_read as disk_blocks_read,
+ heap_blks_hit as buffer_hits,
+ round((100 * heap_blks_hit::float / coalesce(nullif(heap_blks_hit + heap_blks_read, 0), 1))::numeric, 2) as cache_hit_ratio
+ from pg_statio_user_tables
+ order by cache_hit_ratio desc) as foo
+where cache_hit_ratio > 0;
+```
+
+#### Index data in cache per table
+
+```sql
+select *
+from (select relname,
+ idx_blks_read as disk_blocks_read,
+ idx_blks_hit as buffer_hits,
+ round((100 * idx_blks_hit::float / coalesce(nullif(idx_blks_hit + idx_blks_read, 0), 1))::numeric, 2) as cache_hit_ratio
+ from pg_statio_user_tables
+ order by cache_hit_ratio desc) as foo
+where cache_hit_ratio > 0;
+```
+
+#### `bgwriter` frequency
+
+```sql
+select total_checkpoints,
+ seconds_since_start / total_checkpoints / 60 as minutes_between_checkpoints
+from (select extract(epoch from (now() - pg_postmaster_start_time())) as seconds_since_start,
+ (checkpoints_timed + checkpoints_req) as total_checkpoints
+ from pg_stat_bgwriter) as sub;
+```
+
+#### Total sizes of tables
+
+```sql
+select n.nspname as "schema",
+ c.relname as "name",
+ case c.relkind
+ when 'r' then 'table'
+ when 'v' then 'view'
+ when 'm' then 'materialized view'
+ when 'i' then 'index'
+ when 's' then 'sequence'
+ when 't' then 'toast table'
+ when 'f' then 'foreign table'
+ when 'p' then 'partitioned table'
+ when 'i' then 'partitioned index' end as "type",
+ pg_catalog.pg_get_userbyid(c.relowner) as "owner",
+ case c.relpersistence
+ when 'p' then 'permanent'
+ when 't' then 'temporary'
+ when 'u' then 'unlogged' end as "persistence",
+ am.amname as "access method",
+ pg_catalog.pg_size_pretty(pg_catalog.pg_table_size(c.oid)) as "size",
+ pg_catalog.obj_description(c.oid, 'pg_class') as "description"
+from pg_catalog.pg_class c
+ left join pg_catalog.pg_namespace n on n.oid = c.relnamespace
+ left join pg_catalog.pg_am am on am.oid = c.relam
+where c.relkind in ('r', 'p', '')
+ and n.nspname <> 'pg_catalog'
+ and n.nspname !- '-pg_toast'
+ and n.nspname <> 'information_schema'
+ and pg_catalog.pg_table_is_visible(c.oid)
+order by 1, 2;
+```
+
+#### Total sizes of indexes
+
+```sql
+select n.nspname as "schema",
+ c.relname as "name",
+ case c.relkind
+ when 'r' then 'table'
+ when 'v' then 'view'
+ when 'm' then 'materialized view'
+ when 'i' then 'index'
+ when 's' then 'sequence'
+ when 't' then 'toast table'
+ when 'f' then 'foreign table'
+ when 'p' then 'partitioned table'
+ when 'i' then 'partitioned index' end as "type",
+ pg_catalog.pg_get_userbyid(c.relowner) as "owner",
+ c2.relname as "table",
+ case c.relpersistence
+ when 'p' then 'permanent'
+ when 't' then 'temporary'
+ when 'u' then 'unlogged' end as "persistence",
+ am.amname as "access method",
+ pg_catalog.pg_size_pretty(pg_catalog.pg_table_size(c.oid)) as "size",
+ pg_catalog.obj_description(c.oid, 'pg_class') as "description"
+from pg_catalog.pg_class c
+ left join pg_catalog.pg_namespace n on n.oid = c.relnamespace
+ left join pg_catalog.pg_am am on am.oid = c.relam
+ left join pg_catalog.pg_index i on i.indexrelid = c.oid
+ left join pg_catalog.pg_class c2 on i.indrelid = c2.oid
+where c.relkind in ('i', 'i', '')
+ and n.nspname <> 'pg_catalog'
+ and n.nspname !- '-pg_toast'
+ and n.nspname <> 'information_schema'
+ and pg_catalog.pg_table_is_visible(c.oid)
+order by 1, 2;
+```
+
+#### Largest contract instances by payload size
+
+```sql
+select template_fqn,
+ contract_id,
+ octet_length(payload::text)::bigint as size_in_bytes
+from active()
+order by size_in_bytes desc
+limit 10;
+```
+
+#### Largest transactions by payload size
+
+```sql
+select created_at_offset,
+ sum(octet_length(payload::text)::bigint) as size_in_bytes
+from creates()
+group by created_at_offset
+order by size_in_bytes desc
+limit 10;
+```
+
+#### Largest transactions by events count
+
+```sql
+select "offset",
+ count(*) as events_in_tx
+from __events, __transactions
+where __events.tx_ix = __transactions.ix
+group by "offset"
+order by events_in_tx desc
+limit 10;
+```
+
+{/* COPIED_END */}