Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -188,6 +188,10 @@ d_siim = xrv.datasets.SIIM_Pneumothorax_Dataset(imgpath="dicom-images-train/",
d_vin = xrv.datasets.VinDr_Dataset(imgpath=".../train",
csvpath=".../train.csv")

# BRAX, Brazilian labeled chest x-ray dataset. https://doi.org/10.1038/s41597-022-01608-8
d_brax = xrv.datasets.BRAX_Dataset(imgpath="path to brax/1.1.0/images",
csvpath="path to brax/1.1.0/master_spreadsheet_update.csv")

# National Library of Medicine Tuberculosis Datasets. https://www.ncbi.nlm.nih.gov/pmc/articles/PMC4256233/
d_nlmtb = xrv.datasets.NLMTB_Dataset(imgpath="path to MontgomerySet or ChinaSet_AllFiles")

Expand Down
5 changes: 5 additions & 0 deletions docs/source/datasets.rst
Original file line number Diff line number Diff line change
Expand Up @@ -64,6 +64,8 @@ Available Datasets
- CheXpert (Stanford, 224 k images, 14 pathologies)
* - :class:`xrv.datasets.MIMIC_Dataset <xrv.datasets.MIMIC_Dataset>`
- MIMIC-CXR (MIT/PhysioNet, 227 k images)
* - :class:`xrv.datasets.BRAX_Dataset <xrv.datasets.BRAX_Dataset>`
- BRAX (Brazil/PhysioNet, 41 k images)
* - :class:`xrv.datasets.PC_Dataset <xrv.datasets.PC_Dataset>`
- PadChest (Spain, 94 k images)
* - :class:`xrv.datasets.RSNA_Pneumonia_Dataset <xrv.datasets.RSNA_Pneumonia_Dataset>`
Expand Down Expand Up @@ -110,6 +112,9 @@ Dataset Classes
.. autoclass:: xrv.datasets.MIMIC_Dataset
:members: string

.. autoclass:: xrv.datasets.BRAX_Dataset
:members: string

.. autoclass:: xrv.datasets.PC_Dataset
:members: string

Expand Down
78 changes: 78 additions & 0 deletions tests/test_dataloaders.py
Original file line number Diff line number Diff line change
Expand Up @@ -488,3 +488,81 @@ def test_chexlocalize_dataset_requires_segmentation_jsonpath_for_masks(tmp_path)
with pytest.raises(ValueError):
xrv.datasets.CheXlocalize_Dataset(
imgpath=str(tmp_path), csvpath=str(csv_path))


def _make_brax_test_csv(tmp_path):
"""A master_spreadsheet_update.csv-style CSV with one PNG per row under
images/, covering a patient with two images, an uncertain label, an
"85 or more" age, a thin-strip image, a lateral view and a missing view."""
import pandas as pd

pathologies = ["Enlarged Cardiomediastinum", "Cardiomegaly", "Lung Lesion",
"Lung Opacity", "Edema", "Consolidation", "Pneumonia",
"Atelectasis", "Pneumothorax", "Pleural Effusion",
"Pleural Other", "Fracture", "Support Devices"]
rows = [
# patient, view, rows, columns, age, sex, labels
("id_p1", "PA", 2000, 2000, "40", "M", {"No Finding": 1, "Support Devices": 1}),
("id_p1", "PA", 2000, 2000, "40", "M", {"No Finding": 0, "Cardiomegaly": 1}),
("id_p2", "PA", 2000, 1800, "85 or more", "F", {"No Finding": 0, "Pneumonia": -1, "Pleural Effusion": 1}),
("id_p3", "PA", 12, 2000, "50", "M", {"No Finding": 0, "Edema": 1}),
("id_p4", "L", 2000, 2000, "60", "F", {"No Finding": 0, "Fracture": 1}),
("id_p5", np.nan, 2000, 2000, "70", "M", {"No Finding": 1}),
]
records = []
for i, (patient, view, n_rows, n_cols, age, sex, labels) in enumerate(rows):
png_path = f"images/{patient}/Study_1/Series_1/image-{i}.png"
img_file = tmp_path / png_path
img_file.parent.mkdir(parents=True, exist_ok=True)
shutil.copyfile(test_png_img_file, img_file)
record = {"PngPath": png_path, "PatientID": patient, "PatientSex": sex,
"PatientAge": age, "ViewPosition": view, "Rows": n_rows, "Columns": n_cols}
record.update({p: np.nan for p in ["No Finding"] + pathologies})
record.update(labels)
records.append(record)

csv_path = tmp_path / "master_spreadsheet_update.csv"
pd.DataFrame(records).to_csv(csv_path, index=False)
return csv_path


def test_brax_dataset(tmp_path):
csv_path = _make_brax_test_csv(tmp_path)

d = xrv.datasets.BRAX_Dataset(imgpath=str(tmp_path / "images"), csvpath=str(csv_path))

# PA only, strip dropped, one image per patient
assert list(d.csv["PatientID"]) == ["id_p1", "id_p2"]
assert "Effusion" in d.pathologies and "Pleural Effusion" not in d.pathologies

# p1 keeps its first image: No Finding zeroes everything but Support
# Devices, and the second image's Cardiomegaly=1 must not leak in
p1 = dict(zip(d.pathologies, d.labels[0]))
assert p1["Support Devices"] == 1
assert p1["Cardiomegaly"] == 0
assert all(v == 0 for k, v in p1.items() if k != "Support Devices")

p2 = dict(zip(d.pathologies, d.labels[1]))
assert np.isnan(p2["Pneumonia"]) # uncertain (-1) becomes NaN
assert p2["Effusion"] == 1
assert np.isnan(p2["Cardiomegaly"])

assert list(d.csv["age_years"]) == [40, 85]
assert list(d.csv["sex_male"]) == [True, False]

sample = d[0]
assert sample["img"].shape[0] == 1
assert np.array_equal(sample["lab"], d.labels[0], equal_nan=True)


def test_brax_dataset_views_and_aspect_filter(tmp_path):
csv_path = _make_brax_test_csv(tmp_path)
imgpath = str(tmp_path / "images")

d = xrv.datasets.BRAX_Dataset(imgpath=imgpath, csvpath=str(csv_path), views=["*"], unique_patients=False)
assert len(d) == 5
assert "UNKNOWN" in set(d.csv["view"])

d = xrv.datasets.BRAX_Dataset(imgpath=imgpath, csvpath=str(csv_path), views=["*"], unique_patients=False,
max_aspect_ratio=None)
assert len(d) == 6
151 changes: 151 additions & 0 deletions torchxrayvision/datasets.py
Original file line number Diff line number Diff line change
Expand Up @@ -1548,6 +1548,157 @@ def __getitem__(self, idx):
return sample


class BRAX_Dataset(Dataset):
"""BRAX, Brazilian labeled chest X-ray dataset

BRAX contains 40,967 labeled chest radiographs collected at Hospital
Israelita Albert Einstein in São Paulo, Brazil. Labels were
extracted from Brazilian Portuguese radiology reports with a Portuguese
adaptation of the CheXpert labeler, so they use the same encoding as
CheXpert: ``1``, ``0``, ``-1`` (uncertain), or blank. As in
:class:`CheX_Dataset`, ``-1`` is converted to ``NaN`` and "No Finding"
zeroes every other label except Support Devices.

**Pathologies (13):** Atelectasis, Cardiomegaly, Consolidation, Edema,
Effusion, Enlarged Cardiomediastinum, Fracture, Lung Lesion, Lung
Opacity, Pleural Other, Pneumonia, Pneumothorax, Support Devices.

``imgpath`` is the release's ``images/`` folder (the PNG tree) and
``csvpath`` is ``master_spreadsheet_update.csv``. About a quarter of the
images have no ``ViewPosition``; they get the view ``"UNKNOWN"``.

The release contains 12 labeled images whose source DICOM is a thin strip
(aspect ratio 19:1 to 213:1) rather than a chest radiograph; every other
image has an aspect ratio below 9:1. Images whose ``Rows``/``Columns``
aspect ratio exceeds ``max_aspect_ratio`` are dropped. Pass
``max_aspect_ratio=None`` to keep them.

.. note::
BRAX is distributed under the PhysioNet Credentialed Health Data
License 1.5.0. Access requires a credentialed PhysioNet account,
the CITI "Data or Specimens Only Research" training, and signing
the PhysioNet Credentialed Health Data Use Agreement 1.5.0.

Example::

d_brax = xrv.datasets.BRAX_Dataset(
imgpath=".../brax/1.1.0/images",
csvpath=".../brax/1.1.0/master_spreadsheet_update.csv"
)

Citation:
Reis EP, de Paiva JPQ, da Silva MCB, et al.
BRAX, Brazilian labeled chest x-ray dataset.
*Scientific Data* 9, 487 (2022).
https://doi.org/10.1038/s41597-022-01608-8

Dataset website:
https://physionet.org/content/brax/1.1.0/
"""

def __init__(self,
imgpath,
csvpath,
views=["PA"],
transform=None,
data_aug=None,
seed=0,
unique_patients=True,
max_aspect_ratio=10
):

super(BRAX_Dataset, self).__init__()
np.random.seed(seed) # Reset the seed so all runs are the same.

self.pathologies = ["Enlarged Cardiomediastinum",
"Cardiomegaly",
"Lung Opacity",
"Lung Lesion",
"Edema",
"Consolidation",
"Pneumonia",
"Atelectasis",
"Pneumothorax",
"Pleural Effusion",
"Pleural Other",
"Fracture",
"Support Devices"]

self.pathologies = sorted(self.pathologies)

self.imgpath = imgpath
self.csvpath = csvpath
self.transform = transform
self.data_aug = data_aug
self.csv = pd.read_csv(self.csvpath)

self.csv["view"] = self.csv["ViewPosition"]
self.limit_to_selected_views(views)

if max_aspect_ratio is not None:
rows, cols = self.csv["Rows"], self.csv["Columns"]
aspect = np.maximum(rows, cols) / np.minimum(rows, cols)
self.csv = self.csv[aspect <= max_aspect_ratio]

if unique_patients:
# Not groupby().first(): that takes the first non-null value per
# column, mixing labels from different images of the same patient.
self.csv = self.csv.drop_duplicates("PatientID")

self.csv = self.csv.reset_index(drop=True)

# Get our classes.
healthy = self.csv["No Finding"] == 1
labels = []
for pathology in self.pathologies:
if pathology != "Support Devices":
self.csv.loc[healthy, pathology] = 0
labels.append(self.csv[pathology].values)
self.labels = np.asarray(labels).T
self.labels = self.labels.astype(np.float32)

# Make all the -1 values into nans to keep things simple
self.labels[self.labels == -1] = np.nan

# Rename pathologies
self.pathologies = list(np.char.replace(self.pathologies, "Pleural Effusion", "Effusion"))

# add consistent csv values

# patientid
self.csv["patientid"] = self.csv["PatientID"]

# age (5-year age groups, the oldest given as the string "85 or more")
self.csv["age_years"] = pd.to_numeric(self.csv["PatientAge"].replace("85 or more", 85))

# sex
self.csv["sex_male"] = self.csv["PatientSex"] == "M"
self.csv["sex_female"] = self.csv["PatientSex"] == "F"

def string(self):
return self.__class__.__name__ + " num_samples={} views={} data_aug={}".format(len(self), self.views, self.data_aug)

def __len__(self):
return len(self.labels)

def __getitem__(self, idx):
sample = {}
sample["idx"] = idx
sample["lab"] = self.labels[idx]

# CSV paths start with the release's images/ folder, which is imgpath
imgid = self.csv["PngPath"].iloc[idx].replace("images/", "", 1)
img_path = os.path.join(self.imgpath, imgid)
img = imread(img_path)

sample["img"] = normalize(img, maxval=255, reshape=True)

sample = apply_transforms(sample, self.transform)
sample = apply_transforms(sample, self.data_aug)

return sample


class Openi_Dataset(Dataset):
"""OpenI / Indiana University chest X-ray collection

Expand Down
Loading