mirror of
https://github.com/data-privacy-stack/presidio.git
synced 2026-10-02 19:07:58 -05:00
5.9 MiB
5.9 MiB
In [ ]:
!pip install presidio_analyzer presidio_anonymizer presidio_image_redactor
!python -m spacy download en_core_web_lgIn [1]:
import os
import json
import pandas as pd
import pydicom
from presidio_image_redactor import DicomImagePiiVerifyEngineIn [2]:
# Set paths
data_dir = "sample_data"
gt_path = "sample_data/ground_truth.json"In [3]:
# Load ground truth JSON
with open(gt_path) as json_file:
gt = json.load(json_file)
# Get list of files
gt_dicom_files = list(gt.keys())
gt_dicom_filesOut [3]:
['sample_data/0_ORIGINAL.dcm', 'sample_data/1_ORIGINAL.dcm', 'sample_data/2_ORIGINAL.dcm', 'sample_data/3_ORIGINAL.dcm']
In [4]:
dicom_engine = DicomImagePiiVerifyEngine()In [5]:
# Select one file to work with
file_of_interest = gt_dicom_files[0]
gt_file_of_interest = gt[file_of_interest]In [6]:
# Return image to visually inspect
instance = pydicom.dcmread(file_of_interest)
verify_image, ocr_results, analyzer_results = dicom_engine.verify_dicom_instance(instance)In [7]:
def get_PHI_list(PHI: list) -> list:
"""Get list of PHI from ground truth for a single file.
Args:
PHI_dict (list): List of ground truth or detected text PHI.
Return:
PHI_list (list): List of PHI (just text).
"""
PHI_list = []
for item in PHI:
PHI_list.append(item['label'])
return PHI_listIn [ ]:
_, eval_results = dicom_engine.eval_dicom_instance(instance, gt_file_of_interest)In [9]:
print(f"Precision: {eval_results['precision']}")
print(f"Recall: {eval_results['recall']}")
print(f"All Positives: {get_PHI_list(eval_results['all_positives'])}")
print(f"Ground Truth: {get_PHI_list(eval_results['ground_truth'])}")Precision: 1.0 Recall: 1.0 All Positives: ['DAVIDSON', 'DOUGLAS', '[M]', '01.09.2012', '06.16.1976'] Ground Truth: ['DAVIDSON', 'DOUGLAS', '[M]', '01.09.2012', '06.16.1976']
In [10]:
# Initialize lists to turn into results table
list_of_files = gt_dicom_files
list_of_gt = []
list_of_pos = []
list_of_recall = []
list_of_precision = []In [ ]:
for file in gt_dicom_files:
# Setup
ground_truth = gt[file]
instance = pydicom.dcmread(file)
# Evaluate
_, eval_results = dicom_engine.eval_dicom_instance(instance, ground_truth)
# Save results
list_of_gt.append(get_PHI_list(eval_results["ground_truth"]))
list_of_pos.append(get_PHI_list(eval_results["all_positives"]))
list_of_recall.append(eval_results["recall"])
list_of_precision.append(eval_results["precision"])In [12]:
# Organize results into a table
all_results_dict = {
"file": list_of_files,
"ground_truth": list_of_gt,
"all_positives": list_of_pos,
"recall": list_of_recall,
"precision": list_of_precision
}
df_results = pd.DataFrame(all_results_dict)
df_resultsOut [12]:
| file | ground_truth | all_positives | recall | precision | |
|---|---|---|---|---|---|
| 0 | sample_data/0_ORIGINAL.dcm | [DAVIDSON, DOUGLAS, [M], 01.09.2012, 06.16.1976] | [DAVIDSON, DOUGLAS, [M], 01.09.2012, 06.16.1976] | 1.0 | 1.0 |
| 1 | sample_data/1_ORIGINAL.dcm | [MARTIN, CHAD, [U], 01.01.2000] | [MARTIN, CHAD, [U], 01.01.2000] | 1.0 | 1.0 |
| 2 | sample_data/2_ORIGINAL.dcm | [KAUFMAN, SCOTT, [M], 03.09.2012, 07.22.1943] | [KAUFMAN, 07.22.1943, SCOTT, [M], 03.09.2012] | 1.0 | 1.0 |
| 3 | sample_data/3_ORIGINAL.dcm | [MEYER, STEPHANIE, [F], 02.25.2012, 07.16.1953] | [MEYER, STEPHANIE, [F], 02.25.2012, 07.16.1953] | 1.0 | 1.0 |
In [ ]:
# Select file
file_of_interest = gt_dicom_files[3]
gt_file_of_interest = gt[file_of_interest]
instance = pydicom.dcmread(file_of_interest)
# Run evaluation with minimal padding (0 padding not allowed)
_, eval_results = dicom_engine.eval_dicom_instance(instance, gt_file_of_interest, padding_width=1)In [14]:
print(f"Precision: {eval_results['precision']}")
print(f"Recall: {eval_results['recall']}")
print(f"All Positives: {get_PHI_list(eval_results['all_positives'])}")
print(f"Ground Truth: {get_PHI_list(eval_results['ground_truth'])}")Precision: 1.0 Recall: 0.2 All Positives: ['07.16.1953'] Ground Truth: ['MEYER', 'STEPHANIE', '[F]', '02.25.2012', '07.16.1953']