Runnable example · Frontend data pipeline

Score every Ecobalyse activity with VoLCA

A complete pipeline feeding another product: score whole activity catalogues through a running VoLCA server and write the results back in the JSON shape each downstream project expects. One launcher loads databases and methods once; two export scripts read from the same server.

Diagram: ACTIVITY PANEL, Batch scoring, APPLICATION DATA, {, }, Ecobalyse / EcoFoodChoice.
From inventory data to an application input.Batch scoring and application-specific export are distinct operations.

What the scripts do

  • Start a VoLCA server and load every SimaPro CSV database plus the EF 3.1 (adapted) and Biomaps methods with their scoring, from one TOML config
  • Compute EF 3.1 impacts and the ECS single score for every Ecobalyse activity, into Ecobalyse-format JSON
  • Compute the 14 EF 3.1 trigram scores, the regionalized Biomaps biodiversity score, and the EcoFoodChoice single score per activity, into EcoFoodChoice-format JSON

Prerequisites

The Ecobalyse SimaPro CSV exports and an activity catalog from ecobalyse (licensed data, not redistributed here), referenced by volca-ecobalyse.toml. Run the launcher in one terminal, the exports in another, with uv run from the examples project.

Step 1 - launch and load

Starts the engine and loads all databases, methods, and scoring definitions. Both export scripts talk to the server it starts.

uv run ecobalyse/ecobalyse_import.py
ecobalyse_import.py Python 3 · pyvolca
"""VoLCA Ecobalyse import - load all SimaPro CSV databases + EF 3.1 scoring.

Usage:
    cd volca-deploy/volca/examples
    uv run ecobalyse/ecobalyse_import.py [CONFIG.toml]

Serves `volca-ecobalyse.toml` next to this script by default. The sibling
`ecofoodchoice/` example loads the same databases under its own scoring, so it
passes its own config rather than owning a second copy of this launcher.

https://www.volca.run/docs/python/
"""

import subprocess
import sys
import time
import tomllib
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path

from volca import Client, Server, VoLCAError

HERE = Path(__file__).resolve().parent
VOLCA_PUBLIC = HERE.parents[2] / "volca-public"
METHOD = "Environmental Footprint 3.1 (adapted)"


def load_config(path):
    with open(path, "rb") as f:
        return tomllib.load(f)


def resolve_binary(config, config_path):
    """Which engine to launch.

    A config may name one under `server.binary`, but a path written down once
    goes stale at the next GHC or version bump, so it is only honoured while
    the file it names still exists. Otherwise ask cabal what it built last.
    """
    declared = config["server"].get("binary")
    if declared:
        path = (config_path.parent / declared).resolve()
        if path.exists():
            return path
        print(f"  server.binary names {path}, which is gone - asking cabal instead")

    try:
        found = subprocess.run(
            ["cabal", "list-bin", "-O2", "volca"],
            cwd=VOLCA_PUBLIC,
            capture_output=True,
            text=True,
        )
    except FileNotFoundError:  # no cabal on this machine at all
        found = None
    path = Path(found.stdout.strip()) if found and found.returncode == 0 else None
    if path is None or not path.exists():
        raise SystemExit(
            f"No engine to launch: `cabal list-bin -O2 volca` in {VOLCA_PUBLIC} "
            f"answered nothing usable. Build it first, or point `server.binary` "
            f"in {config_path.name} at one."
        )
    return path


def main():
    config_path = Path(sys.argv[1]).resolve() if len(sys.argv) > 1 else HERE / "volca-ecobalyse.toml"
    if not config_path.is_file():
        raise SystemExit(f"No such config: {config_path}")
    config = load_config(config_path)

    databases = config["databases"]
    methods = config["methods"]

    print("=" * 70)
    print("  VoLCA Demo - SimaPro CSV databases + EF 3.1 scoring")
    print("=" * 70)

    # --- Start server ---
    print(f"\n► Starting VoLCA server (port {config['server']['port']})...")
    srv = Server(config=str(config_path), binary=str(resolve_binary(config, config_path)))
    srv.start(wait_timeout=30)
    client = Client(base_url=srv.base_url, password=srv.password)
    print(f"  Server ready at {srv.base_url}")

    # --- Show configured databases ---
    print(f"\n► {len(databases)} databases configured:")
    for db in databases:
        deps = db.get("depends", [])
        dep_info = f" (depends on {', '.join(deps)})" if deps else ""
        print(f"  · {db['displayName']}{dep_info}")

    # --- Show configured methods ---
    print(f"\n► {len(methods)} method(s) configured:")
    for m in methods:
        scoring_sets = m.get("scoring", [])
        scoring_info = f" - scoring: {', '.join(s['name'] for s in scoring_sets)}" if scoring_sets else ""
        print(f"  · {m['name']}{scoring_info}")

    # --- Load databases ---
    print("\n► Loading databases into memory...")
    for db in databases:
        name, display = db["name"], db["displayName"]
        print(f"  Loading {display}...", end=" ", flush=True)
        t = time.perf_counter()
        try:
            client.load_database(name)
            print(f"done ({time.perf_counter() - t:.1f}s)")
        except VoLCAError as e:
            print(f"FAILED ({e})")

    # --- Search for a test activity ---
    test_db = "agribalyse-3-2"
    query = "mozzarella"
    print(f"\n► Searching '{query}' in {test_db}...")
    activity = client.use(test_db).search_activities(name=query, limit=1)[0]
    print(f"  Found: {activity.activity_name}")
    print(f"  Process ID: {activity.process_id}")

    # --- Compute impacts ---
    print(f"\n► Computing LCIA impacts ({METHOD})...")
    result = client.use(test_db).get_impacts_batch(activity.process_id, collection=METHOD)

    print(f"  {len(result.results)} impact categories computed\n")
    print(f"  {'Category':<50s} {'Score':>12s}  Unit")
    print(f"  {'-' * 76}")
    for cat in result.results:
        print(f"  {cat.category:<50s} {cat.score:12.4e}  {cat.unit}")

    # --- Scoring results ---
    if result.scoring_results:
        print("\n► Scoring results (formula-based single scores):")
        for set_name, scores in result.scoring_results.items():
            unit = result.scoring_units.get(set_name, "Pt")
            for score_name, value in scores.items():
                label = set_name if len(scores) == 1 else f"{set_name}.{score_name}"
                print(f"  {label} = {value:.2f} {unit}")

    # --- Pre-warm method tables for every loaded database (parallel) ---
    # First /impacts hit on a (db, method-collection) pair pays the
    # regionalized weight precomputation up-front. Doing it here, in
    # parallel across DBs, moves that cost OUT of any later batch run -
    # ecobalyse_export.py then measures only the steady-state per-pid
    # characterization.
    print(f"\n► Warming up {len(databases)} databases (one /impacts call each, parallel)...")
    warm_start = time.perf_counter()

    def _warm_one(db_cfg):
        name, display = db_cfg["name"], db_cfg["displayName"]
        t0 = time.perf_counter()
        c = client.use(name)
        # Grab any activity to hand to /impacts; the work is per-(db, method),
        # not per-pid, so the choice of pid doesn't matter.
        results = c.search_activities(limit=1)
        if not results:
            return display, "no activities", time.perf_counter() - t0
        try:
            c.score_activities([results[0].process_id], collection=METHOD, top_flows=0)
            status = "ok"
        except VoLCAError as e:
            status = str(e)
        return display, status, time.perf_counter() - t0

    with ThreadPoolExecutor(max_workers=len(databases)) as pool:
        futures = {pool.submit(_warm_one, db): db for db in databases}
        for fut in as_completed(futures):
            display, status, dt = fut.result()
            print(f"  [{dt:5.1f}s] {display}: {status}")
    print(f"  Warmup wall-clock: {time.perf_counter() - warm_start:.1f}s")

    print()


if __name__ == "__main__":
    main()

Step 2 - export in Ecobalyse format

uv run ecobalyse/ecobalyse_export.py
ecobalyse_export.py Python 3 · pyvolca
"""VoLCA Ecobalyse export - compute impacts for all Ecobalyse activities.

Reads activities from ecobalyse-data4/activities.json, connects to a running
VoLCA server (started by ecobalyse_import.py), computes EF 3.1 impacts + ECS
score for each process, and writes results in Ecobalyse JSON format.

Usage:
    cd volca-deploy/volca/examples
    uv run ecobalyse/ecobalyse_import.py   # start server + load databases
    uv run ecobalyse/ecobalyse_export.py   # compute impacts (in another terminal)

https://www.volca.run/docs/python/
"""

import argparse
import json
import sys
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path

from volca import Client, VoLCAError

HERE = Path(__file__).resolve().parent
OUTPUT_JSON = HERE / "ecobalyse_activities.json"
BASE_URL = "http://localhost:8080"
METHOD = "Environmental Footprint 3.1 (adapted)"
WORKERS = 8

# Source field in lci_catalog activity JSON → DB slug loaded in the running server.
# "Ecobalyse" and "Custom" sources are virtual aggregates without a backing
# SimaPro/EcoSpold export - they're skipped silently here.
SOURCE_TO_DB = {
    "Agribalyse 3.2": "agribalyse-3-2",
    "Ecoinvent 3.9.1": "ecoinvent-3-9-1-adapted",
    "Ecoinvent 3.11": "ecoinvent-3-11-adapted",
    "Ginko 2025": "ginko-2025-v2",
    "PastoEco": "pastoeco",
    "Woolmark": "woolmark",
    "WFLDB": "wfldb",
}

# EF 3.1 impact category name → Ecobalyse trigram
CATEGORY_TO_TRIGRAM = {
    "Acidification": "acd",
    "Climate change": "cch",
    "Eutrophication, freshwater": "fwe",
    "Eutrophication, marine": "swe",
    "Eutrophication, terrestrial": "tre",
    "Ecotoxicity, freshwater - part 1": "etf1",
    "Ecotoxicity, freshwater - part 2": "etf2",
    "Human toxicity, cancer": "htc",
    "Human toxicity, non-cancer": "htn",
    "Ionising radiation": "ior",
    "Land use": "ldu",
    "Ozone depletion": "ozd",
    "Particulate matter": "pma",
    "Photochemical ozone formation": "pco",
    "Resource use, fossils": "fru",
    "Resource use, minerals and metals": "mru",
    "Water use": "wtu",
}


def check_server(client):
    """Verify the VoLCA server is running and return loaded database names."""
    try:
        client.get_version()
    except Exception:
        print("ERROR: VoLCA server not running. Start it first with:")
        print("  uv run ecobalyse_import.py")
        sys.exit(1)

    return {db.name for db in client.list_databases() if db.status == "loaded"}


def resolve_pid(client, db_name, activity):
    """Resolve one activity name to a processId. Returns (activity, pid_or_None, status)."""
    try:
        found = client.use(db_name).search_activities(
            name=activity["activityName"], limit=1
        ).results
    except VoLCAError:
        return activity, None, "failed"
    if not found:
        return activity, None, "not_found"
    return activity, found[0].process_id, "ok"


def extract_impacts(batch_result):
    """Map an LCIABatchResult to a {trigram: score} dict."""
    impacts = {}
    for cat in batch_result.results:
        trigram = CATEGORY_TO_TRIGRAM.get(cat.category)
        if trigram:
            impacts[trigram] = cat.score
    for set_name, scores in batch_result.scoring_results.items():
        if set_name == "ECS":
            impacts["ecs"] = scores.get("total", 0)
        elif set_name == "ecobalyse":
            for score_name, value in scores.items():
                impacts[score_name] = value
    return impacts


def batch_impacts(client, db_name, pid_to_activity):
    """Score all pids for this DB in one MUMPS multi-RHS solve.
    Returns {pid: impacts_dict}. Missing pids → empty dict entries filled by caller.

    Sends `top_flows=0` to skip the O(|inventory|×|methods|) contribution walk
    we don't consume; raw scores + scoring sets are unaffected."""
    if not pid_to_activity:
        return {}
    scored = client.use(db_name).score_activities(
        list(pid_to_activity.keys()), collection=METHOD, top_flows=0
    )
    return {sa.process_id: extract_impacts(sa.impacts) for sa in scored.results}


def build_record(activity, impacts):
    return {
        "activityName": activity["activityName"],
        "comment": activity.get("comment", ""),
        "displayName": activity.get("displayName", ""),
        "impacts": {k: format_number(v) for k, v in sorted(impacts.items())},
        "location": activity.get("location", ""),
        "massPerUnit": activity.get("massPerUnit"),
        "scopes": activity.get("scopes", []),
        "source": activity["source"],
        "unit": activity.get("unit", ""),
        "waste": activity.get("waste", 0),
    }


def format_number(v):
    """Format like Ecobalyse: 5 significant digits, use int for zero.

    Tolerates non-numeric server values (e.g. an error string surfaced by a
    scoring formula that referenced a Left-tainted regionalized score). We
    pass them through unchanged so the export still emits a record; the
    downstream consumer can spot and triage non-numeric entries.
    """
    if v is None:
        return None
    if not isinstance(v, (int, float)):
        return v
    if v == 0:
        return 0
    return float(f"{v:.5g}")


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument(
        "--only-db",
        metavar="NAME",
        action="append",
        default=[],
        help="Restrict to one TOML db name (repeatable, e.g. --only-db ecoinvent-3.11)",
    )
    parser.add_argument(
        "--warmup",
        action="store_true",
        help="Issue a GET /impacts for one activity per DB first to pay factorization outside the timed section",
    )
    parser.add_argument(
        "--lci-catalog",
        metavar="DIR",
        type=Path,
        default=Path("lci_catalog"),
        help="Path to the ecobalyse-data lci_catalog directory (one JSON file per activity under <db>/)",
    )
    args = parser.parse_args()
    if not args.lci_catalog.is_dir():
        parser.error(f"--lci-catalog: no such directory: {args.lci_catalog}")

    # Read input activities - one JSON file per activity under lci_catalog/<db>/
    print(f"► Reading activities from {args.lci_catalog}/<db>/*.json...")
    activities = [
        json.load(open(p))
        for p in sorted(args.lci_catalog.glob("*/*.json"))
    ]
    computable = [
        a
        for a in activities
        if a.get("activityName") and a.get("source") in SOURCE_TO_DB
    ]
    if args.only_db:
        allowed = set(args.only_db)
        computable = [a for a in computable if SOURCE_TO_DB[a["source"]] in allowed]
        print(
            f"  --only-db filter: keeping {len(computable)} activities across {len(allowed)} databases"
        )
    skipped_no_name = sum(1 for a in activities if not a.get("activityName"))
    skipped_unknown_source = sum(
        1
        for a in activities
        if a.get("activityName") and a.get("source") not in SOURCE_TO_DB
    )
    print(f"  {len(activities)} total, {len(computable)} computable")
    print(
        f"  Skipped: {skipped_no_name} without activityName, {skipped_unknown_source} unknown source"
    )

    # Check server is running
    print(f"\n► Connecting to VoLCA server at {BASE_URL}...")
    client = Client(base_url=BASE_URL)
    loaded_dbs = check_server(client)
    print(
        f"  Connected. {len(loaded_dbs)} databases loaded: {', '.join(sorted(loaded_dbs))}"
    )

    # Warn about missing databases
    needed_dbs = {SOURCE_TO_DB[a["source"]] for a in computable}
    missing = needed_dbs - loaded_dbs
    if missing:
        print(f"  WARNING: databases not loaded: {', '.join(sorted(missing))}")
        computable = [a for a in computable if SOURCE_TO_DB[a["source"]] in loaded_dbs]
        print(f"  Reduced to {len(computable)} computable activities")

    # Build (db_name, activity) pairs
    tasks = [(SOURCE_TO_DB[a["source"]], a) for a in computable]

    results = []
    not_found = []
    failed = []
    total = len(tasks)
    total_start = time.perf_counter()

    # Phase 1: resolve names → processIds in parallel (light I/O-bound work)
    print(f"\n► Resolving {total} processIds ({WORKERS} parallel searches)...")
    phase1_start = time.perf_counter()
    by_db = {}  # db_name -> {pid: activity}
    with ThreadPoolExecutor(max_workers=WORKERS) as pool:
        futures = [pool.submit(resolve_pid, client, db, act) for db, act in tasks]
        for done, fut in enumerate(as_completed(futures), start=1):
            activity, pid, status = fut.result()
            display = activity.get("displayName", activity.get("activityName", "?"))
            db_name = SOURCE_TO_DB[activity["source"]]
            if status == "ok":
                by_db.setdefault(db_name, {})[pid] = activity
            elif status == "not_found":
                not_found.append((db_name, display, activity["activityName"]))
            else:
                failed.append((db_name, display))
            if done % 200 == 0 or done == total:
                print(f"  [{done}/{total}]")
    print(f"  Resolved in {time.perf_counter() - phase1_start:.1f}s")

    # Optional warmup: force MUMPS factorization + method flow-mapping cache per DB
    # so Phase 2 measures only the steady-state per-activity cost.
    if args.warmup:
        print(
            f"\n► Warming {len(by_db)} databases (GET /impacts on first pid of each)..."
        )
        warm_start = time.perf_counter()
        for db_name, pid_map in by_db.items():
            first_pid = next(iter(pid_map))
            t = time.perf_counter()
            client.use(db_name).get_impacts_batch(first_pid, collection=METHOD)
            print(f"  [{db_name}] warmup {time.perf_counter() - t:.2f}s")
        print(f"  Warmup total {time.perf_counter() - warm_start:.1f}s")

    # Phase 2: one batch-impacts call per DB, run in parallel across DBs.
    # Different DBs use different MUMPS handles so concurrent solves don't
    # contend; wall-clock becomes max(per-DB) instead of sum.
    print(f"\n► Batch impacts across {len(by_db)} databases (parallel)...")
    phase2_start = time.perf_counter()

    def run_batch(db_name, pid_map):
        t0 = time.perf_counter()
        impacts_by_pid = batch_impacts(client, db_name, pid_map)
        return db_name, pid_map, impacts_by_pid, time.perf_counter() - t0

    with ThreadPoolExecutor(max_workers=max(1, len(by_db))) as pool:
        futures = [pool.submit(run_batch, db, pm) for db, pm in by_db.items()]
        for fut in as_completed(futures):
            db_name, pid_map, impacts_by_pid, elapsed = fut.result()
            print(
                f"  [{db_name}] {len(impacts_by_pid)}/{len(pid_map)} in {elapsed:.1f}s"
                f" ({1000 * elapsed / max(1, len(pid_map)):.0f}ms/process)"
            )
            for pid, activity in pid_map.items():
                impacts = impacts_by_pid.get(pid)
                if impacts is None:
                    failed.append((
                        db_name,
                        activity.get("displayName", activity["activityName"]),
                    ))
                else:
                    results.append(build_record(activity, impacts))
    print(f"  Phase 2 wall-clock: {time.perf_counter() - phase2_start:.1f}s")

    done = len(results) + len(not_found) + len(failed)
    total_time = time.perf_counter() - total_start

    # Write output
    print(f"\n► Writing {len(results)} activities to {OUTPUT_JSON.name}...")
    with open(OUTPUT_JSON, "w", encoding="utf-8") as f:
        json.dump(results, f, indent=2, ensure_ascii=False)
        f.write("\n")

    # Summary
    print(f"\n{'=' * 60}")
    print(
        f"  Results:  {len(results)} computed, {len(not_found)} not found, {len(failed)} failed"
    )
    if done > 0:
        mean_ms = 1000 * total_time / done
        print(f"  Timing:   {total_time:.1f}s total, {mean_ms:.0f}ms mean per process")
    if not_found:
        print(f"\n  Not found ({len(not_found)}):")
        for db, display, name in not_found[:10]:
            print(f"    [{db}] {display}")
        if len(not_found) > 10:
            print(f"    ... and {len(not_found) - 10} more")
    print()


if __name__ == "__main__":
    main()

Or - export in EcoFoodChoice format

uv run ecofoodchoice/ecofoodchoice_export.py
ecofoodchoice_export.py Python 3 · pyvolca
"""VoLCA EcoFoodChoice export - compute EFC + Biomaps impacts.

Reads activities from ecobalyse-data/lci_catalog/<db>/*.json, connects to a
running VoLCA server, computes EF 3.1 impacts + ECS + EFC + regionalized
Biomaps for each process, and writes a JSON file with one record per activity
carrying:

  - the 14 PEF/EF 3.1 trigram scores (raw characterized values)
  - the regionalized Biomaps biodiversity-loss potential ("bio")
  - the EFC single-score total ("efc")

The Biomaps method is regionalized (CFs vary by activity location). VoLCA
resolves the per-activity location to its CF cell through the geography
hierarchy built into the engine; a TOML names `geographies` only to replace
that hierarchy with a file of its own.

Usage:
    cd volca-deploy/volca/examples
    uv run ecobalyse/ecobalyse_import.py ecofoodchoice/volca-ecofoodchoice.toml
    uv run ecofoodchoice/ecofoodchoice_export.py   # in another terminal

https://www.volca.run/docs/python/
"""

import argparse
import json
import sys
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path

from volca import Client

HERE = Path(__file__).resolve().parent
OUTPUT_JSON = HERE / "ecofoodchoice_activities.json"
BASE_URL = "http://localhost:8080"
METHOD = "Environmental Footprint 3.1 (adapted)"
WORKERS = 8

# Source field in lci_catalog activity JSON → DB slug loaded in the running server.
# "Ecobalyse" and "Custom" sources are virtual aggregates without a backing
# SimaPro/EcoSpold export - they're skipped silently here.
SOURCE_TO_DB = {
    "Agribalyse 3.2": "agribalyse-3-2",
    "Ecoinvent 3.9.1": "ecoinvent-3-9-1-adapted",
    "Ecoinvent 3.11": "ecoinvent-3-11-adapted",
    "Ginko 2025": "ginko-2025-v2",
    "PastoEco": "pastoeco",
    "Woolmark": "woolmark",
    "WFLDB": "wfldb",
}

# Impact category name → EFC short code (trigrams kept in sync with EF 3.1).
# Biomaps is a single regionalized category; EFC is the formula-based total.
CATEGORY_TO_TRIGRAM = {
    "Acidification": "acd",
    "Climate change": "cch",
    "Eutrophication, freshwater": "fwe",
    "Eutrophication, marine": "swe",
    "Eutrophication, terrestrial": "tre",
    "Ecotoxicity, freshwater - inorganics": "etfi",
    "Ecotoxicity, freshwater - organics - p.1": "etfo1",
    "Ecotoxicity, freshwater - organics - p.2": "etfo2",
    "Ionising radiation": "ior",
    "Land use": "ldu",
    "Ozone depletion": "ozd",
    "Particulate matter": "pma",
    "Photochemical ozone formation": "pco",
    "Resource use, fossils": "fru",
    "Resource use, minerals and metals": "mru",
    "Water use": "wtu",
    "Biomaps - Biodiversity Loss Potential": "bio",
}


def check_server(client):
    """Verify the VoLCA server is running and return loaded database names."""
    try:
        client.get_version()
    except Exception:
        print("ERROR: VoLCA server not running. Start it first with:")
        print("  uv run ecobalyse/ecobalyse_import.py ecofoodchoice/volca-ecofoodchoice.toml")
        sys.exit(1)

    return {db.name for db in client.list_databases() if db.status == "loaded"}


def resolve_db(client, db_name, activities):
    """Resolve one database's catalogue names to processIds, exactly.

    The catalogue's `activityName` is the SimaPro *reference product*, marker
    letter and all ("... {FR} U"). The engine keeps that string on the product
    and strips the marker from the activity name, so matching it against
    activity names misses most of the catalogue - and a substring search always
    ranks something first, which is how a miss turns into another process's
    scores filed under the name asked for. Matching it against products is
    exact and unique.

    Returns ({pid: activity}, [(db, display, reason)]): a name that matches no
    product, or several, lands in the second list rather than being guessed at.
    """
    by_product = {a["activityName"]: a for a in activities}
    matches = client.use(db_name).resolve_activities(
        by_product, by="product", workers=WORKERS
    )
    resolved, unresolved = {}, []
    for name, found in matches.items():
        activity = by_product[name]
        if len(found) == 1:
            resolved[found[0].process_id] = activity
        else:
            reason = "no match" if not found else f"{len(found)} matches"
            unresolved.append((db_name, activity.get("displayName") or name, reason))
    return resolved, unresolved


def extract_impacts(batch_result):
    """Map an LCIABatchResult to an EFC-shaped {trigram: score} dict.

    Carries the 14 PEF-mapped EF 3.1 categories + the regionalized Biomaps
    score + the EFC formula total. The raw inorganic / organics-p.1 / -p.2
    Ecotoxicity, freshwater categories ship as-is (etfi / etfo1 / etfo2) so
    downstream consumers can re-aggregate the way they want; EFC's own
    `organics = etfo1 + etfo2` happens server-side via the scoring formula
    and is exposed in `scoring_results`.
    """
    impacts = {}
    for cat in batch_result.results:
        trigram = CATEGORY_TO_TRIGRAM.get(cat.category)
        if trigram:
            impacts[trigram] = cat.score
    # A scoring set that answered without a `total` has failed, and writing a
    # zero would publish "no impact" for it. Leave the key out instead, so a
    # consumer sees a missing score rather than a free product.
    for set_name, key in (("EFC", "efc"), ("ECS", "ecs"), ("PEF", "pef")):
        total = batch_result.scoring_results.get(set_name, {}).get("total")
        if total is not None:
            impacts[key] = total
    return impacts


def batch_impacts(client, db_name, pid_to_activity):
    """Score all pids for this DB in one MUMPS multi-RHS solve.
    Returns {pid: impacts_dict}. Missing pids → empty dict entries filled by caller.

    Sends `top_flows=0` to skip the O(|inventory|×|methods|) contribution walk
    we don't consume; raw scores + scoring sets are unaffected."""
    if not pid_to_activity:
        return {}
    scored = client.use(db_name).score_activities(
        list(pid_to_activity.keys()), collection=METHOD, top_flows=0
    )
    return {sa.process_id: extract_impacts(sa.impacts) for sa in scored.results}


def build_record(activity, impacts):
    return {
        "activityName": activity["activityName"],
        "comment": activity.get("comment", ""),
        "displayName": activity.get("displayName", ""),
        "impacts": {k: format_number(v) for k, v in sorted(impacts.items())},
        "location": activity.get("location", ""),
        "massPerUnit": activity.get("massPerUnit"),
        "scopes": activity.get("scopes", []),
        "source": activity["source"],
        "unit": activity.get("unit", ""),
        "waste": activity.get("waste", 0),
    }


def format_number(v):
    """Format like Ecobalyse: 5 significant digits, use int for zero.

    Tolerates non-numeric server values (e.g. an error string surfaced by a
    scoring formula that referenced a Left-tainted regionalized score). We
    pass them through unchanged so the export still emits a record; the
    downstream consumer can spot and triage non-numeric entries.
    """
    if v is None:
        return None
    if not isinstance(v, (int, float)):
        return v
    if v == 0:
        return 0
    return float(f"{v:.5g}")


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument(
        "--only-db",
        metavar="NAME",
        action="append",
        default=[],
        help="Restrict to one TOML db name (repeatable, e.g. --only-db ecoinvent-3.11)",
    )
    parser.add_argument(
        "--warmup",
        action="store_true",
        help="Issue a GET /impacts for one activity per DB first to pay factorization outside the timed section",
    )
    parser.add_argument(
        "--lci-catalog",
        metavar="DIR",
        type=Path,
        default=Path("lci_catalog"),
        help="Path to the ecobalyse-data lci_catalog directory (one JSON file per activity under <db>/)",
    )
    parser.add_argument(
        "--url",
        default=BASE_URL,
        help=f"VoLCA server base URL (default {BASE_URL})",
    )
    args = parser.parse_args()
    if not args.lci_catalog.is_dir():
        parser.error(f"--lci-catalog: no such directory: {args.lci_catalog}")

    # Read input activities - one JSON file per activity under lci_catalog/<db>/
    print(f"► Reading activities from {args.lci_catalog}/<db>/*.json...")
    activities = [
        json.load(open(p))
        for p in sorted(args.lci_catalog.glob("*/*.json"))
    ]
    computable = [
        a
        for a in activities
        if a.get("activityName") and a.get("source") in SOURCE_TO_DB
    ]
    if args.only_db:
        allowed = set(args.only_db)
        computable = [a for a in computable if SOURCE_TO_DB[a["source"]] in allowed]
        print(
            f"  --only-db filter: keeping {len(computable)} activities across {len(allowed)} databases"
        )
    skipped_no_name = sum(1 for a in activities if not a.get("activityName"))
    skipped_unknown_source = sum(
        1
        for a in activities
        if a.get("activityName") and a.get("source") not in SOURCE_TO_DB
    )
    print(f"  {len(activities)} total, {len(computable)} computable")
    print(
        f"  Skipped: {skipped_no_name} without activityName, {skipped_unknown_source} unknown source"
    )

    # Check server is running
    print(f"\n► Connecting to VoLCA server at {args.url}...")
    client = Client(base_url=args.url)
    loaded_dbs = check_server(client)
    print(
        f"  Connected. {len(loaded_dbs)} databases loaded: {', '.join(sorted(loaded_dbs))}"
    )

    # A missing database is not a reason to carry on: the committed file covers
    # every configured one, and finishing without them would overwrite it with a
    # shorter file that looks complete. Note a partially linked database does not
    # count as loaded - the engine refuses to score one.
    needed_dbs = {SOURCE_TO_DB[a["source"]] for a in computable}
    missing = needed_dbs - loaded_dbs
    if missing and not args.only_db:
        raise SystemExit(
            f"  Databases not loaded: {', '.join(sorted(missing))}\n"
            f"  Load them before exporting, or pass --only-db to write a subset "
            f"on purpose."
        )
    if missing:
        print(f"  WARNING: databases not loaded: {', '.join(sorted(missing))}")
        computable = [a for a in computable if SOURCE_TO_DB[a["source"]] in loaded_dbs]
        print(f"  Reduced to {len(computable)} computable activities")

    # Group the activities by the database that holds them
    per_db = {}
    for a in computable:
        per_db.setdefault(SOURCE_TO_DB[a["source"]], []).append(a)

    results = []
    unresolved_names = []
    failed = []
    total = len(computable)
    total_start = time.perf_counter()

    # Phase 1: resolve names → processIds, one concurrent batch per database
    print(f"\n► Resolving {total} processIds across {len(per_db)} databases...")
    phase1_start = time.perf_counter()
    by_db = {}  # db_name -> {pid: activity}
    for db_name, acts in per_db.items():
        resolved, unresolved = resolve_db(client, db_name, acts)
        by_db[db_name] = resolved
        unresolved_names.extend(unresolved)
        print(f"  [{db_name}] {len(resolved)}/{len(acts)} resolved")
    print(f"  Resolved in {time.perf_counter() - phase1_start:.1f}s")

    # Optional warmup: force MUMPS factorization + method flow-mapping cache per DB
    # so Phase 2 measures only the steady-state per-activity cost.
    if args.warmup:
        print(
            f"\n► Warming {len(by_db)} databases (GET /impacts on first pid of each)..."
        )
        warm_start = time.perf_counter()
        for db_name, pid_map in by_db.items():
            first_pid = next(iter(pid_map))
            t = time.perf_counter()
            client.use(db_name).get_impacts_batch(first_pid, collection=METHOD)
            print(f"  [{db_name}] warmup {time.perf_counter() - t:.2f}s")
        print(f"  Warmup total {time.perf_counter() - warm_start:.1f}s")

    # Phase 2: one batch-impacts call per DB, run in parallel across DBs.
    # Different DBs use different MUMPS handles so concurrent solves don't
    # contend; wall-clock becomes max(per-DB) instead of sum.
    print(f"\n► Batch impacts across {len(by_db)} databases (parallel)...")
    phase2_start = time.perf_counter()

    def run_batch(db_name, pid_map):
        t0 = time.perf_counter()
        impacts_by_pid = batch_impacts(client, db_name, pid_map)
        return db_name, pid_map, impacts_by_pid, time.perf_counter() - t0

    with ThreadPoolExecutor(max_workers=max(1, len(by_db))) as pool:
        futures = [pool.submit(run_batch, db, pm) for db, pm in by_db.items()]
        for fut in as_completed(futures):
            db_name, pid_map, impacts_by_pid, elapsed = fut.result()
            print(
                f"  [{db_name}] {len(impacts_by_pid)}/{len(pid_map)} in {elapsed:.1f}s"
                f" ({1000 * elapsed / max(1, len(pid_map)):.0f}ms/process)"
            )
            for pid, activity in pid_map.items():
                impacts = impacts_by_pid.get(pid)
                if impacts is None:
                    failed.append((
                        db_name,
                        activity.get("displayName", activity["activityName"]),
                    ))
                else:
                    results.append(build_record(activity, impacts))
    print(f"  Phase 2 wall-clock: {time.perf_counter() - phase2_start:.1f}s")

    done = len(results) + len(unresolved_names) + len(failed)
    total_time = time.perf_counter() - total_start

    # Write output, in a stable order: the batches come back as they finish, and
    # a file whose records move around between runs diffs as a rewrite even when
    # almost nothing changed.
    results.sort(key=lambda r: (r["source"], r["activityName"]))
    print(f"\n► Writing {len(results)} activities to {OUTPUT_JSON.name}...")
    with open(OUTPUT_JSON, "w", encoding="utf-8") as f:
        json.dump(results, f, indent=2, ensure_ascii=False)
        f.write("\n")

    # Summary
    print(f"\n{'=' * 60}")
    print(
        f"  Results:  {len(results)} computed, {len(unresolved_names)} not resolved, {len(failed)} failed"
    )
    if done > 0:
        mean_ms = 1000 * total_time / done
        print(f"  Timing:   {total_time:.1f}s total, {mean_ms:.0f}ms mean per process")
    if unresolved_names:
        # Printed in full: every one of these is an activity the catalogue asks
        # for and the export does not carry, so a cut-off list would understate
        # what is missing from the file.
        print(f"\n  Not resolved ({len(unresolved_names)}):")
        for db, display, reason in unresolved_names:
            print(f"    [{db}] {display} - {reason}")
    print()


if __name__ == "__main__":
    main()

Feeding your own application?

The same pattern - one launcher, exports shaped for your consumer - works for any frontend that needs environmental scores.