diff --git a/science-containers/Dockerfiles/lsst-science-platform/Dockerfile b/science-containers/Dockerfiles/lsst-science-platform/Dockerfile index 9647456..abbe82a 100755 --- a/science-containers/Dockerfiles/lsst-science-platform/Dockerfile +++ b/science-containers/Dockerfiles/lsst-science-platform/Dockerfile @@ -44,7 +44,16 @@ RUN mkdir /skaha COPY src/startup.sh /skaha/startup.sh RUN chmod +x /skaha/startup.sh COPY src/launch_firefly_on_canfar.py /skaha/launch_firefly_on_canfar.py +COPY src/populate_discovery.py /skaha/populate_discovery.py +RUN chmod +x /skaha/populate_discovery.py COPY src/jupyter /usr/bin/jupyter +# cadc_dataset_map.yaml is NOT copied into the image; it is deployed to +# /arc/projects/LSST/ via Argo CD (see arc-projects-LSST/README.md). + +# Writable Nublado discovery/token paths used by lsst.rsp.RSPDiscovery. +# Skaha sessions run as arbitrary UIDs, so these must be world-writable. +RUN mkdir -p /etc/nublado/discovery /etc/nublado/secrets \ + && chmod 1777 /etc/nublado /etc/nublado/discovery /etc/nublado/secrets # Some items to make this a better experience on CANFAR RUN /skaha/startup.sh pip install cadctap cadcdata vos canfar safir rsp-jupyter-extensions lsdb diff --git a/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/README.md b/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/README.md new file mode 100644 index 0000000..15cae0c --- /dev/null +++ b/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/README.md @@ -0,0 +1,93 @@ +# ARC project files for LSST on CANFAR + +Files in this directory are **not** baked into the LSST science-platform +container image. They are the source of truth for configuration that must +live on shared ARC storage at: + +```text +/arc/projects/LSST/ +``` + +Session startup scripts (for example `startup.sh` in the sciplat image) +read these paths at runtime so operators can update service mappings +without rebuilding images. + +## Files + +| File in this directory | Deployed path | Used by | +| --- | --- | --- | +| `cadc_dataset_map.yaml` | `/arc/projects/LSST/cadc_dataset_map.yaml` | `populate_discovery.py` → `/etc/nublado/discovery/v1.json` for `lsst.rsp.RSPDiscovery` | +| `cadc_repositories.yaml` | `/arc/projects/LSST/cadc_repositories.yaml` | `DAF_BUTLER_REPOSITORY_INDEX` in `startup.sh` (also at `https://www.canfar.net/storage/arc/file/projects/LSST/cadc_repositories.yaml`) | + +## Argo CD deployment + +These files should be synced by the platform Argo CD configuration in +[opencadc/science-platform](https://github.com/opencadc/science-platform) +(or the LSST project’s ARC provisioning workflow), **not** by the Docker +build for this image. + +Recommended pattern: + +1. Keep the canonical copies in this directory (this repo). +2. Add an Argo CD / GitOps sync that publishes them to + `/arc/projects/LSST/` on the CANFAR storage backend used by Skaha + sessions. +3. After changing `cadc_dataset_map.yaml`, new notebook sessions pick up + the update on next start (discovery JSON is regenerated each startup). +4. After changing `cadc_repositories.yaml`, new sessions see updated Butler + labels via `DAF_BUTLER_REPOSITORY_INDEX`. + +Do **not** `COPY` these files into the Dockerfile. The container only +ships `populate_discovery.py` and env wiring that expect the ARC paths. + +## Local testing + +```bash +python science-containers/Dockerfiles/lsst-science-platform/src/populate_discovery.py \ + --map science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_dataset_map.yaml \ + --dry-run +``` + +## Dataset map notes + +- Dataset labels (`dp1`, `dp2`, `dp02`, `prompt`, …) are arguments to + `RSPDiscovery("dp1")`. +- Service values may be: + - a CADC IVOID from + [resource-caps](https://ws.cadc-ccda.hia-iha.nrc-cnrc.gc.ca/reg/resource-caps) + - a literal `https://...` URL (Rubin mirrors) + - a `{url, versions}` dict +- `dp1` / `dp2`: CANFAR YouCAT, GMS, SIA, SODA cutout, DataLink, plus + Butler configs on `ws-uv.canfar.net`. HiPS points at Rubin. +- `dp02` / `prompt`: discovery entries that point at `data.lsst.cloud`. + +### Dual-token authentication + +`RSPDiscovery` sends **one** bearer token to every service URL in a dataset. +CANFAR (CADC) and Rubin (Gafaelfawr) tokens are not interchangeable. + +| Dataset | Default session token (CADC) | Rubin token required | +| --- | --- | --- | +| `dp1`, `dp2` | tap, gms, sia, cutout, datalink | hips (Rubin URL) | +| `dp02`, `prompt` | will not work | all services | + +```python +import os +from lsst.rsp import RSPDiscovery + +# CANFAR services (session token from /etc/nublado/secrets/token) +canfar = RSPDiscovery("dp2") +tap = canfar.get_tap_client() + +# Rubin-mirrored discovery (explicit Rubin token) +rubin = RSPDiscovery("dp02", token=os.environ["RUBIN_TOKEN"]) +hips_url = rubin.get_service_url("hips") +``` + +Create Rubin tokens via the RSP token UI: +https://rsp.lsst.io/guides/auth/creating-user-tokens.html + +## Butler repository index notes + +- Keys are Butler repository labels (`dp1`, `canfar-dp1`, `rubin-dp1`, …). +- Values are Butler configuration YAML URLs for CANFAR or Rubin endpoints. diff --git a/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_dataset_map.yaml b/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_dataset_map.yaml new file mode 100644 index 0000000..9cc7f9f --- /dev/null +++ b/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_dataset_map.yaml @@ -0,0 +1,140 @@ +# Map RSP dataset labels to CADC/IVOA resources and optional literal URLs. +# +# Deployed location (outside the container image): +# /arc/projects/LSST/cadc_dataset_map.yaml +# +# At session start, populate_discovery.py reads this file and writes +# /etc/nublado/discovery/v1.json for lsst.rsp.RSPDiscovery. +# +# Service values may be: +# - ivo://... → resolve via resource-caps + VOSI capabilities (CANFAR/CADC) +# - https://... → literal URL (used to mirror Rubin services into discovery) +# - {url, versions} dict for fully specified literal entries +# +# AUTH WARNING +# ------------ +# RSPDiscovery attaches a single bearer token to every service URL for a +# dataset. CANFAR (CADC) tokens and Rubin (Gafaelfawr) tokens are different. +# - dp1 / dp2: default session token is the CADC token from startup.sh. +# CANFAR-resolved services (tap, gms, sia, cutout, datalink) work with it. +# hips entries below point at Rubin and will NOT authenticate with that +# CADC token; call them only with an explicit Rubin token, or use the +# Rubin-mirrored datasets (dp02 / prompt) instead. +# - dp02 / prompt: URLs point at data.lsst.cloud. Use a Rubin token: +# RSPDiscovery("dp02", token=os.environ["RUBIN_TOKEN"]) +# Do not expect the default CANFAR ACCESS_TOKEN / NUBLADO_TOKEN to work. +# +# Source of truth: science-containers/.../arc-projects-LSST/ +# Sync to ARC via Argo CD (see README.md in that directory). + +resource_caps_url: https://ws.cadc-ccda.hia-iha.nrc-cnrc.gc.ca/reg/resource-caps +discovery_path: /etc/nublado/discovery/v1.json +environment_name: canfar + +# RSPDiscovery service name → IVOA capability standardID(s) to match +# inside a resource's VOSI capabilities document (IVOID entries only). +service_standard_ids: + tap: + - ivo://ivoa.net/std/TAP + sia: + - ivo://ivoa.net/std/SIA#query-2.0 + - ivo://ivoa.net/std/SIA + datalink: + - ivo://ivoa.net/std/DataLink#links-1.1 + - ivo://ivoa.net/std/DataLink + cutout: + - ivo://ivoa.net/std/SODA#sync-1.0 + - ivo://ivoa.net/std/SODA + gms: + - ivo://ivoa.net/std/GMS#search-1.0 + - ivo://ivoa.net/std/GMS#search-0.1 + +datasets: + # --- CANFAR-authenticated datasets (default session CADC token) --- + dp1: + description: >- + Data Preview 1 on CANFAR (YouCAT + CAOM ops + CADC GMS). HiPS URL is + Rubin's and needs a separate Rubin token. + docs_url: https://dp1.lsst.io/ + butler_config: https://ws-uv.canfar.net/lsst/api/butler/repo/dp1/butler.yaml + services: + tap: ivo://cadc.nrc.ca/youcat + gms: ivo://cadc.nrc.ca/gms + sia: ivo://cadc.nrc.ca/sia + cutout: ivo://cadc.nrc.ca/caom2ops + datalink: ivo://cadc.nrc.ca/caom2ops + hips: + url: https://data.lsst.cloud/api/hips/v2/dp1/list + versions: + hips-list-1.0: + url: https://data.lsst.cloud/api/hips/v2/dp1/list + + dp2: + description: >- + Data Preview 2 on CANFAR (YouCAT + CAOM ops + CADC GMS). HiPS URL is + Rubin's and needs a separate Rubin token. + docs_url: https://dp2.lsst.io/ + butler_config: https://ws-uv.canfar.net/lsst/api/butler/repo/dp2/butler.yaml + services: + tap: ivo://cadc.nrc.ca/youcat + gms: ivo://cadc.nrc.ca/gms + sia: ivo://cadc.nrc.ca/sia + cutout: ivo://cadc.nrc.ca/caom2ops + datalink: ivo://cadc.nrc.ca/caom2ops + hips: + url: https://data.lsst.cloud/api/hips/v2/dp2/list + versions: + hips-list-1.0: + url: https://data.lsst.cloud/api/hips/v2/dp2/list + + # --- Rubin-mirrored datasets (require a Rubin Gafaelfawr token) --- + dp02: + description: >- + DP0.2 discovery entries pointing at Rubin data.lsst.cloud services. + Authenticate with a Rubin token, not the CANFAR session token. + docs_url: https://dp0-2.lsst.io/ + butler_config: https://data.lsst.cloud/api/butler/repo/dp02/butler.yaml + services: + tap: + url: https://data.lsst.cloud/api/tap + versions: + tables: + url: https://data.lsst.cloud/api/tap/tables + sia: + url: https://data.lsst.cloud/api/sia/dp02 + versions: + sia-query-2.0: + url: https://data.lsst.cloud/api/sia/dp02/query + cutout: + url: https://data.lsst.cloud/api/cutout + versions: + soda-async-1.0: + url: https://data.lsst.cloud/api/cutout/jobs + soda-sync-1.0: + url: https://data.lsst.cloud/api/cutout/sync + datalink: + url: https://data.lsst.cloud/api/datalink + versions: + datalink-links-1.1: + url: https://data.lsst.cloud/api/datalink/links + gms: + url: https://data.lsst.cloud/auth/gms + versions: + gms-search-1.0: + url: https://data.lsst.cloud/auth/gms + hips: + url: https://data.lsst.cloud/api/hips/v2/dp02/list + versions: + hips-list-1.0: + url: https://data.lsst.cloud/api/hips/v2/dp02/list + + prompt: + description: >- + Rubin prompt-product discovery (alerts). Requires a Rubin token. + services: + alerts: https://data.lsst.cloud/api/alerts + gms: + url: https://data.lsst.cloud/auth/gms + versions: + gms-search-1.0: + url: https://data.lsst.cloud/auth/gms diff --git a/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_repositories.yaml b/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_repositories.yaml new file mode 100644 index 0000000..ceab6e3 --- /dev/null +++ b/science-containers/Dockerfiles/lsst-science-platform/arc-projects-LSST/cadc_repositories.yaml @@ -0,0 +1,18 @@ +# Butler repository index for CANFAR LSST sessions. +# +# Deployed location (outside the container image): +# /arc/projects/LSST/cadc_repositories.yaml +# +# Referenced by startup.sh as DAF_BUTLER_REPOSITORY_INDEX +# (also served at +# https://www.canfar.net/storage/arc/file/projects/LSST/cadc_repositories.yaml). +# +# Source of truth: science-containers/.../arc-projects-LSST/ +# Sync to ARC via Argo CD (see README.md in this directory). + +dp1: "https://ws-uv.canfar.net/lsst/api/butler/repo/dp1/butler.yaml" +dp2: "https://ws-uv.canfar.net/lsst/api/butler/repo/dp2/butler.yaml" +canfar-dp1: "https://ws-uv.canfar.net/lsst/api/butler/repo/dp1/butler.yaml" +canfar-dp2: "https://ws-uv.canfar.net/lsst/api/butler/repo/dp2/butler.yaml" +rubin-dp1: "https://data.lsst.cloud/api/butler/repo/dp1/butler.yaml" +rubin-dp2: "https://data.lsst.cloud/api/butler/repo/dp2/butler.yaml" diff --git a/science-containers/Dockerfiles/lsst-science-platform/src/populate_discovery.py b/science-containers/Dockerfiles/lsst-science-platform/src/populate_discovery.py new file mode 100644 index 0000000..ed3c0b7 --- /dev/null +++ b/science-containers/Dockerfiles/lsst-science-platform/src/populate_discovery.py @@ -0,0 +1,359 @@ +#!/usr/bin/env python3 +"""Populate Nublado discovery JSON from CADC IVOA resource-caps. + +Reads a dataset map, resolves IVOID services via the CADC registry +resource-caps file (or accepts literal service URL entries), and writes +``/etc/nublado/discovery/v1.json`` for ``lsst.rsp.RSPDiscovery``. + +Service map values may be: + +* an IVOID (``ivo://...``) resolved through resource-caps + VOSI capabilities +* a literal ``https://...`` URL (e.g. Rubin services mirrored into CANFAR discovery) +* a dict with ``url`` and optional ``versions`` (passed through as-is) +""" + +from __future__ import annotations + +import argparse +import json +import sys +import urllib.error +import urllib.request +import xml.etree.ElementTree as ET +from pathlib import Path +from typing import Any + +try: + import yaml +except ImportError: # pragma: no cover - sciplat images include PyYAML + yaml = None + +# Shared ARC config (not baked into the image). Synced via Argo CD; see +# arc-projects-LSST/README.md in this Dockerfile tree. +DEFAULT_MAP = Path("/arc/projects/LSST/cadc_dataset_map.yaml") +USER_AGENT = "canfar-populate-discovery/1.0" + +# Extra version keys to attach when the matching VOSI capability exists. +SERVICE_VERSION_STANDARDS: dict[str, list[tuple[str, tuple[str, ...]]]] = { + "tap": [ + ( + "tables", + ( + "ivo://ivoa.net/std/VOSI#tables-1.1", + "ivo://ivoa.net/std/VOSI#tables", + ), + ), + ], + "sia": [ + ("sia-query-2.0", ("ivo://ivoa.net/std/SIA#query-2.0",)), + ], + "datalink": [ + ( + "datalink-links-1.1", + ( + "ivo://ivoa.net/std/DataLink#links-1.1", + "ivo://ivoa.net/std/DataLink#links-1.0", + ), + ), + ], + "cutout": [ + ("soda-sync-1.0", ("ivo://ivoa.net/std/SODA#sync-1.0",)), + ("soda-async-1.0", ("ivo://ivoa.net/std/SODA#async-1.0",)), + ], + "gms": [ + ( + "gms-search-1.0", + ( + "ivo://ivoa.net/std/GMS#search-1.0", + "ivo://ivoa.net/std/GMS#search-0.1", + ), + ), + ], +} + + +def _load_map(path: Path) -> dict[str, Any]: + text = path.read_text(encoding="utf-8") + if path.suffix.lower() in {".yaml", ".yml"}: + if yaml is None: + raise RuntimeError( + "PyYAML is required to read cadc_dataset_map.yaml; " + "install pyyaml or provide a JSON map" + ) + data = yaml.safe_load(text) + else: + data = json.loads(text) + if not isinstance(data, dict): + raise ValueError(f"dataset map root must be a mapping: {path}") + return data + + +def _http_get(url: str, timeout: float = 30.0) -> bytes: + request = urllib.request.Request( + url, + headers={"User-Agent": USER_AGENT, "Accept": "*/*"}, + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + return response.read() + + +def parse_resource_caps(text: str) -> dict[str, str]: + """Parse CADC resource-caps into IVOID → capabilities URL.""" + mapping: dict[str, str] = {} + for raw_line in text.splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + if "=" not in line: + continue + ivid, caps_url = line.split("=", 1) + ivid = ivid.strip() + caps_url = caps_url.strip() + if ivid and caps_url: + mapping[ivid] = caps_url + return mapping + + +def _local(tag: str) -> str: + if "}" in tag: + return tag.rsplit("}", 1)[-1] + return tag + + +def _find_access_url(capability: ET.Element) -> str | None: + """Return the preferred accessURL for a VOSI capability.""" + interfaces = [c for c in capability if _local(c.tag) == "interface"] + ordered = sorted( + interfaces, + key=lambda el: 0 if el.attrib.get("role") == "std" else 1, + ) + for interface in ordered: + for child in interface: + if _local(child.tag) != "accessURL": + continue + url = (child.text or "").strip() + if url: + return url + return None + + +def parse_capabilities(xml_bytes: bytes) -> dict[str, dict[str, str]]: + """Map capability standardID → {url}.""" + root = ET.fromstring(xml_bytes) + by_standard: dict[str, dict[str, str]] = {} + for capability in root.iter(): + if _local(capability.tag) != "capability": + continue + standard_id = capability.attrib.get("standardID") + if not standard_id: + continue + access_url = _find_access_url(capability) + if access_url: + by_standard[standard_id] = {"url": access_url} + return by_standard + + +def _match_standard( + capabilities: dict[str, dict[str, str]], candidates: list[str] | tuple[str, ...] +) -> tuple[str, dict[str, str]] | None: + for candidate in candidates: + if candidate in capabilities: + return candidate, capabilities[candidate] + for standard_id, info in capabilities.items(): + if standard_id == candidate or standard_id.startswith( + candidate + "#" + ): + return standard_id, info + return None + + +def resolve_ivoa_service( + capabilities: dict[str, dict[str, str]], + service_name: str, + standard_ids: list[str], +) -> dict[str, Any] | None: + """Build an RSPDiscovery service entry from VOSI capabilities.""" + matched = _match_standard(capabilities, standard_ids) + if not matched: + return None + _, info = matched + entry: dict[str, Any] = {"url": info["url"]} + + versions: dict[str, dict[str, str]] = {} + for version_name, version_ids in SERVICE_VERSION_STANDARDS.get( + service_name, [] + ): + version_match = _match_standard(capabilities, version_ids) + if version_match: + _, version_info = version_match + versions[version_name] = {"url": version_info["url"]} + if versions: + entry["versions"] = versions + + return entry + + +def _normalize_literal_service(spec: Any) -> dict[str, Any]: + """Accept a URL string or {url, versions} dict for non-IVOA entries.""" + if isinstance(spec, str): + return {"url": spec} + if isinstance(spec, dict) and "url" in spec: + entry: dict[str, Any] = {"url": str(spec["url"])} + if versions := spec.get("versions"): + entry["versions"] = { + name: {"url": str(info["url"])} + if isinstance(info, dict) + else {"url": str(info)} + for name, info in versions.items() + } + return entry + raise ValueError(f"invalid literal service spec: {spec!r}") + + +def resolve_service_spec( + *, + dataset: str, + service_name: str, + spec: Any, + resource_caps: dict[str, str], + standard_map: dict[str, list[str]], + caps_cache: dict[str, dict[str, dict[str, str]]], +) -> dict[str, Any]: + """Resolve one service map entry to an RSPDiscovery service object.""" + if isinstance(spec, dict) and "url" in spec: + return _normalize_literal_service(spec) + + if not isinstance(spec, str): + raise ValueError( + f"dataset {dataset!r} service {service_name!r}: " + f"expected IVOID, URL, or {{url: ...}} dict, got {spec!r}" + ) + + if spec.startswith("http://") or spec.startswith("https://"): + return _normalize_literal_service(spec) + + if not spec.startswith("ivo://"): + raise ValueError( + f"dataset {dataset!r} service {service_name!r}: " + f"unrecognized service reference {spec!r}" + ) + + if spec not in resource_caps: + raise KeyError( + f"dataset {dataset!r} service {service_name!r}: " + f"IVOID {spec!r} not found in resource-caps" + ) + caps_url = resource_caps[spec] + if caps_url not in caps_cache: + caps_cache[caps_url] = parse_capabilities(_http_get(caps_url)) + capabilities = caps_cache[caps_url] + + standard_ids = standard_map.get(service_name, []) + if not standard_ids: + raise KeyError( + f"no service_standard_ids configured for {service_name!r}" + ) + entry = resolve_ivoa_service(capabilities, service_name, standard_ids) + if entry is None: + raise RuntimeError( + f"dataset {dataset!r} service {service_name!r}: " + f"no capability matching {standard_ids} at {caps_url}" + ) + return entry + + +def build_discovery(config: dict[str, Any]) -> dict[str, Any]: + resource_caps_url = config["resource_caps_url"] + standard_map: dict[str, list[str]] = config.get("service_standard_ids", {}) + datasets_cfg: dict[str, Any] = config.get("datasets", {}) + + caps_text = _http_get(resource_caps_url).decode("utf-8", errors="replace") + resource_caps = parse_resource_caps(caps_text) + caps_cache: dict[str, dict[str, dict[str, str]]] = {} + + datasets: dict[str, Any] = {} + for dataset, meta in datasets_cfg.items(): + services_cfg = meta.get("services") or {} + services: dict[str, Any] = {} + for service_name, spec in services_cfg.items(): + services[service_name] = resolve_service_spec( + dataset=dataset, + service_name=service_name, + spec=spec, + resource_caps=resource_caps, + standard_map=standard_map, + caps_cache=caps_cache, + ) + + dataset_entry: dict[str, Any] = {} + if butler := meta.get("butler_config"): + dataset_entry["butler_config"] = butler + if services: + dataset_entry["services"] = services + datasets[dataset] = dataset_entry + + discovery: dict[str, Any] = {"datasets": datasets, "services": {}} + if env_name := config.get("environment_name"): + discovery["environment_name"] = env_name + return discovery + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--map", + type=Path, + default=DEFAULT_MAP, + help=f"dataset map YAML/JSON (default: {DEFAULT_MAP})", + ) + parser.add_argument( + "--output", + type=Path, + default=None, + help="override discovery output path from the map file", + ) + parser.add_argument( + "--dry-run", + action="store_true", + help="print discovery JSON to stdout instead of writing a file", + ) + args = parser.parse_args(argv) + + try: + config = _load_map(args.map) + discovery = build_discovery(config) + except FileNotFoundError as exc: + print( + f"populate_discovery: dataset map not found: {exc.filename}\n" + " Expected /arc/projects/LSST/cadc_dataset_map.yaml " + "(deploy via Argo CD; see arc-projects-LSST/README.md).", + file=sys.stderr, + ) + return 1 + except ( + OSError, + ValueError, + KeyError, + RuntimeError, + urllib.error.URLError, + ET.ParseError, + ) as exc: + print(f"populate_discovery: {exc}", file=sys.stderr) + return 1 + + payload = json.dumps(discovery, indent=2, sort_keys=True) + "\n" + if args.dry_run: + sys.stdout.write(payload) + return 0 + + output = args.output or Path( + config.get("discovery_path", "/etc/nublado/discovery/v1.json") + ) + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(payload, encoding="utf-8") + print(f"populate_discovery: wrote {output}", file=sys.stderr) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/science-containers/Dockerfiles/lsst-science-platform/src/rsp_notice b/science-containers/Dockerfiles/lsst-science-platform/src/rsp_notice index c96e158..93e249c 100644 --- a/science-containers/Dockerfiles/lsst-science-platform/src/rsp_notice +++ b/science-containers/Dockerfiles/lsst-science-platform/src/rsp_notice @@ -4,17 +4,32 @@ Rubin Observatory Science Platform Notebook Aspect running and configured for CANFAR. If you have issues please contact: support@canfar.net -Rubin TAP services access requires an access token. Set the -environment variable ACCESS_TOKEN to the value of your Rubin Access -Token. +Service discovery is populated at session start from +/arc/projects/LSST/cadc_dataset_map.yaml via the CADC resource-caps +registry into /etc/nublado/discovery/v1.json. Your CADC access token +is published where RSPDiscovery expects it +(/etc/nublado/secrets/token, NUBLADO_TOKEN, ACCESS_TOKEN). + +Example (CANFAR YouCAT / CAOM services): + + from lsst.rsp import RSPDiscovery + discovery = RSPDiscovery("dp2") + tap = discovery.get_tap_client() + +Rubin-mirrored datasets (dp02, prompt) and HiPS URLs on dp1/dp2 point +at data.lsst.cloud and need a separate Rubin token, e.g.: + + import os + rubin = RSPDiscovery("dp02", token=os.environ["RUBIN_TOKEN"]) + +Create Rubin tokens at: + https://rsp.lsst.io/guides/auth/creating-user-tokens.html -Rubin Access Tokens can be craeted using these instructions: - https://rsp.lsst.io/guides/auth/creating-user-tokens.html Find useful documentation for the software and Notebook Aspect at: https://pipelines.lsst.io https://rsp.lsst.io + https://repertoire.lsst.io/user-guide/nublado.html The Rubin Observatory Science Pipelines environment is in: /opt/lsst/software/stack - diff --git a/science-containers/Dockerfiles/lsst-science-platform/src/startup.sh b/science-containers/Dockerfiles/lsst-science-platform/src/startup.sh index 593aaca..af97b86 100755 --- a/science-containers/Dockerfiles/lsst-science-platform/src/startup.sh +++ b/science-containers/Dockerfiles/lsst-science-platform/src/startup.sh @@ -7,7 +7,7 @@ export EXTERNAL_INSTANCE_URL="https://data.lsst.cloud" export DAF_BUTLER_REPOSITORY_INDEX="https://www.canfar.net/storage/arc/file/projects/LSST/cadc_repositories.yaml" export TMPDIR="/tmp" -# use the cadc remote butler which is connected to Storage Inventory +# use the cadc remote butler which is connected to Storage Inventory export DAF_BUTLER_SERVER_GAFAELFAWR_URL=DISABLED export DAF_BUTLER_SERVER_AUTHENTICATION=cadc @@ -15,11 +15,29 @@ export DAF_BUTLER_SERVER_AUTHENTICATION=cadc . /etc/profile . /opt/lsst/software/stack/loadLSST.bash +# Authenticate to CADC and publish the token where RSPDiscovery looks for it: +# 1) /etc/nublado/secrets/token (preferred) +# 2) NUBLADO_TOKEN / ACCESS_TOKEN environment variables +mkdir -p /etc/nublado/secrets /etc/nublado/discovery [ -f ~/.ssl/cadcproxy.pem ] || canfar auth login -[ -f ~/.ssl/cadcproxy.pem ] && export CADC_TOKEN=$(curl -E ~/.ssl/cadcproxy.pem "https://ws-cadc.canfar.net/ac/authorize?response_type=token") +if [ -f ~/.ssl/cadcproxy.pem ]; then + TOKEN=$(curl -fsS -E ~/.ssl/cadcproxy.pem \ + "https://ws-cadc.canfar.net/ac/authorize?response_type=token") + (umask 077; printf '%s\n' "$TOKEN" > /etc/nublado/secrets/token) + export NUBLADO_TOKEN="$TOKEN" + export ACCESS_TOKEN="$TOKEN" + export CADC_TOKEN="$TOKEN" +fi + +# Resolve CADC IVOA resource-caps into RSPDiscovery's Nublado discovery file. +# Dataset map lives on shared ARC storage (Argo CD), not in the image. +python /skaha/populate_discovery.py \ + --map /arc/projects/LSST/cadc_dataset_map.yaml \ + || echo "WARNING: failed to populate /etc/nublado/discovery/v1.json" >&2 + export fireflyURLLab="$(python /skaha/launch_firefly_on_canfar.py | grep https)" export FIREFLY_URL="${fireflyURLLab}" -# start a server that can start firefly for the user +# start a server that can start firefly for the user # [ -d /opt/csp_firefly ] && (cd /opt/csp_firefly; uvicorn main:app --port 8080 >& http_log.txt &) exec $@