add fix
This commit is contained in:
parent
15c319eb9e
commit
56c07edbbd
Binary file not shown.
488
dicom_indexer.py
488
dicom_indexer.py
@ -19,9 +19,17 @@ MongoDB connection config:
|
|||||||
MONGO_URI=mongodb://localhost:27017
|
MONGO_URI=mongodb://localhost:27017
|
||||||
MONGO_DB=dicom_index
|
MONGO_DB=dicom_index
|
||||||
MONGO_COLLECTION=files
|
MONGO_COLLECTION=files
|
||||||
|
MONGO_USERNAME=myuser
|
||||||
|
MONGO_PASSWORD=mypassword
|
||||||
|
|
||||||
CLI flags --mongo-uri/--db/--collection are still available and take
|
MONGO_USERNAME/MONGO_PASSWORD are optional — skip them if your MongoDB
|
||||||
priority over the .env file if you ever need a one-off override.
|
has no auth, or if you'd rather embed credentials directly in
|
||||||
|
MONGO_URI (e.g. mongodb://user:pass@host:27017). Using the separate
|
||||||
|
fields avoids having to URL-encode special characters in the password.
|
||||||
|
|
||||||
|
CLI flags --mongo-uri/--db/--collection/--mongo-username/--mongo-password
|
||||||
|
are still available and take priority over the .env file if you ever
|
||||||
|
need a one-off override.
|
||||||
|
|
||||||
Usage:
|
Usage:
|
||||||
# Build/update an index from a folder (uses .env for Mongo connection)
|
# Build/update an index from a folder (uses .env for Mongo connection)
|
||||||
@ -192,9 +200,9 @@ SKIP_EXTENSIONS = {
|
|||||||
# extensions (or no extension) to be opened at all.
|
# extensions (or no extension) to be opened at all.
|
||||||
DICOM_EXTENSIONS = {".dcm", ".dicom", ".dic", ".ima", ".img", ".dcm30"}
|
DICOM_EXTENSIONS = {".dcm", ".dicom", ".dic", ".ima", ".img", ".dcm30"}
|
||||||
|
|
||||||
DEFAULT_MONGO_URI = "mongodb://root:AtHEntRyPrOchite@172.16.44.35:27017/"
|
DEFAULT_MONGO_URI = "mongodb://localhost:27017"
|
||||||
DEFAULT_DB_NAME = "rsabhk_simrs"
|
DEFAULT_DB_NAME = "dicom_index"
|
||||||
DEFAULT_COLLECTION_NAME = "dicom-index"
|
DEFAULT_COLLECTION_NAME = "files"
|
||||||
|
|
||||||
|
|
||||||
def format_dicom_date(value):
|
def format_dicom_date(value):
|
||||||
@ -255,10 +263,16 @@ def extract_metadata(filepath):
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# MongoDB helpers
|
# MongoDB helpers
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
def get_collection(mongo_uri, db_name, collection_name):
|
def get_collection(mongo_uri, db_name, collection_name, username=None, password=None):
|
||||||
"""Connect to MongoDB and return the target collection, with indexes set up."""
|
"""Connect to MongoDB and return the target collection, with indexes set up."""
|
||||||
|
client_kwargs = {"serverSelectionTimeoutMS": 5000}
|
||||||
|
if username:
|
||||||
|
client_kwargs["username"] = username
|
||||||
|
if password:
|
||||||
|
client_kwargs["password"] = password
|
||||||
|
|
||||||
try:
|
try:
|
||||||
client = MongoClient(mongo_uri, serverSelectionTimeoutMS=5000)
|
client = MongoClient(mongo_uri, **client_kwargs)
|
||||||
client.admin.command("ping")
|
client.admin.command("ping")
|
||||||
except PyMongoError as exc:
|
except PyMongoError as exc:
|
||||||
print(f"Error: could not connect to MongoDB at '{mongo_uri}': {exc}", file=sys.stderr)
|
print(f"Error: could not connect to MongoDB at '{mongo_uri}': {exc}", file=sys.stderr)
|
||||||
@ -344,7 +358,7 @@ def cmd_scan(args):
|
|||||||
print(f"Error: '{folder}' is not a valid folder.", file=sys.stderr)
|
print(f"Error: '{folder}' is not a valid folder.", file=sys.stderr)
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
coll = get_collection(args.mongo_uri, args.db, args.collection)
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
if args.rescan:
|
if args.rescan:
|
||||||
already_indexed = set()
|
already_indexed = set()
|
||||||
@ -429,7 +443,7 @@ DISPLAY_LIMIT = 200 # cap rows kept in memory for screen display
|
|||||||
|
|
||||||
|
|
||||||
def cmd_search(args):
|
def cmd_search(args):
|
||||||
coll = get_collection(args.mongo_uri, args.db, args.collection)
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
mongo_filter = {}
|
mongo_filter = {}
|
||||||
for flag, column in SEARCH_FLAG_TO_COLUMN.items():
|
for flag, column in SEARCH_FLAG_TO_COLUMN.items():
|
||||||
@ -437,6 +451,17 @@ def cmd_search(args):
|
|||||||
if value:
|
if value:
|
||||||
mongo_filter[column] = {"$regex": value, "$options": "i"}
|
mongo_filter[column] = {"$regex": value, "$options": "i"}
|
||||||
|
|
||||||
|
if args.accession_missing:
|
||||||
|
if args.accession:
|
||||||
|
print("Note: --accession-missing overrides --accession; ignoring the --accession value.")
|
||||||
|
mongo_filter.pop("AccessionNumber", None)
|
||||||
|
# Matches AccessionNumber that is missing, null, or an empty/whitespace-only string.
|
||||||
|
mongo_filter["$or"] = [
|
||||||
|
{"AccessionNumber": {"$exists": False}},
|
||||||
|
{"AccessionNumber": None},
|
||||||
|
{"AccessionNumber": {"$regex": r"^\s*$"}},
|
||||||
|
]
|
||||||
|
|
||||||
if args.query:
|
if args.query:
|
||||||
searchable_columns = [c for c, _ in DICOM_FIELDS] + ["FileName", "FilePath"]
|
searchable_columns = [c for c, _ in DICOM_FIELDS] + ["FileName", "FilePath"]
|
||||||
mongo_filter["$or"] = [
|
mongo_filter["$or"] = [
|
||||||
@ -470,7 +495,7 @@ def cmd_search(args):
|
|||||||
out_f.close()
|
out_f.close()
|
||||||
|
|
||||||
display_columns = args.columns.split(",") if args.columns else [
|
display_columns = args.columns.split(",") if args.columns else [
|
||||||
"PatientName", "PatientID", "StudyDate", "StudyDescription",
|
"PatientName", "PatientID", "AccessionNumber", "StudyDate", "StudyDescription",
|
||||||
"Modality", "SeriesDescription", "SeriesNumber", "InstanceNumber",
|
"Modality", "SeriesDescription", "SeriesNumber", "InstanceNumber",
|
||||||
"FileName",
|
"FileName",
|
||||||
]
|
]
|
||||||
@ -488,7 +513,7 @@ def cmd_search(args):
|
|||||||
# Subcommand: summary
|
# Subcommand: summary
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
def cmd_summary(args):
|
def cmd_summary(args):
|
||||||
coll = get_collection(args.mongo_uri, args.db, args.collection)
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
total = coll.count_documents({})
|
total = coll.count_documents({})
|
||||||
if total == 0:
|
if total == 0:
|
||||||
@ -503,7 +528,7 @@ def cmd_summary(args):
|
|||||||
{"$group": {"_id": "$Modality", "count": {"$sum": 1}}},
|
{"$group": {"_id": "$Modality", "count": {"$sum": 1}}},
|
||||||
{"$sort": {"count": -1}},
|
{"$sort": {"count": -1}},
|
||||||
]
|
]
|
||||||
modality_counts = list(coll.aggregate(modality_pipeline))
|
modality_counts = list(coll.aggregate(modality_pipeline, allowDiskUse=True))
|
||||||
|
|
||||||
group_pipeline = [
|
group_pipeline = [
|
||||||
{"$group": {
|
{"$group": {
|
||||||
@ -514,7 +539,7 @@ def cmd_summary(args):
|
|||||||
"count": {"$sum": 1},
|
"count": {"$sum": 1},
|
||||||
}},
|
}},
|
||||||
]
|
]
|
||||||
group_counts = list(coll.aggregate(group_pipeline))
|
group_counts = list(coll.aggregate(group_pipeline, allowDiskUse=True))
|
||||||
|
|
||||||
print("\n--- Index Summary ---")
|
print("\n--- Index Summary ---")
|
||||||
print(f"Total files indexed: {total}")
|
print(f"Total files indexed: {total}")
|
||||||
@ -538,7 +563,7 @@ def cmd_summary(args):
|
|||||||
# Subcommand: export
|
# Subcommand: export
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
def cmd_export(args):
|
def cmd_export(args):
|
||||||
coll = get_collection(args.mongo_uri, args.db, args.collection)
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
total = coll.count_documents({})
|
total = coll.count_documents({})
|
||||||
if total == 0:
|
if total == 0:
|
||||||
@ -554,6 +579,340 @@ def cmd_export(args):
|
|||||||
print(f"Exported {total} row(s) to: {args.output}")
|
print(f"Exported {total} row(s) to: {args.output}")
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Subcommand: dedupe
|
||||||
|
#
|
||||||
|
# Since FilePath is the unique key for indexing, the index can never have
|
||||||
|
# two documents with the identical path. But the same DICOM instance can
|
||||||
|
# legitimately end up indexed at more than one path — e.g. a file copied to
|
||||||
|
# two folders, or re-exported by the PACS to a new location. Those share
|
||||||
|
# the same SOPInstanceUID (globally unique per DICOM standard), which is
|
||||||
|
# what this command uses to detect true duplicates.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
def cmd_dedupe(args):
|
||||||
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
|
pipeline = [
|
||||||
|
{"$match": {"SOPInstanceUID": {"$nin": [None, ""]}}},
|
||||||
|
{"$group": {
|
||||||
|
"_id": "$SOPInstanceUID",
|
||||||
|
"count": {"$sum": 1},
|
||||||
|
"docs": {"$push": {"FilePath": "$FilePath", "IndexedAt": "$IndexedAt"}},
|
||||||
|
}},
|
||||||
|
{"$match": {"count": {"$gt": 1}}},
|
||||||
|
]
|
||||||
|
groups = list(coll.aggregate(pipeline, allowDiskUse=True))
|
||||||
|
|
||||||
|
if not groups:
|
||||||
|
print("No duplicate DICOM instances found (checked by SOPInstanceUID).")
|
||||||
|
return
|
||||||
|
|
||||||
|
total_dupe_files = sum(g["count"] for g in groups)
|
||||||
|
to_remove = 0
|
||||||
|
|
||||||
|
print(f"Found {len(groups)} DICOM instance(s) indexed under more than one file path "
|
||||||
|
f"({total_dupe_files} file entries total).")
|
||||||
|
if args.dry_run:
|
||||||
|
print("(--dry-run given: reporting only, NOT deleting anything from MongoDB)")
|
||||||
|
print(f"Keep strategy: {args.keep} entry per duplicate group "
|
||||||
|
f"(files that no longer exist on disk are never kept if a valid copy is available).\n")
|
||||||
|
|
||||||
|
paths_to_delete = []
|
||||||
|
shown = 0
|
||||||
|
|
||||||
|
for g in groups:
|
||||||
|
docs = g["docs"]
|
||||||
|
existing = [d for d in docs if os.path.isfile(d["FilePath"])]
|
||||||
|
candidates = existing if existing else docs
|
||||||
|
|
||||||
|
candidates_sorted = sorted(
|
||||||
|
candidates, key=lambda d: d.get("IndexedAt") or "", reverse=(args.keep == "newest")
|
||||||
|
)
|
||||||
|
keep = candidates_sorted[0]
|
||||||
|
remove = [d for d in docs if d["FilePath"] != keep["FilePath"]]
|
||||||
|
|
||||||
|
paths_to_delete.extend(d["FilePath"] for d in remove)
|
||||||
|
to_remove += len(remove)
|
||||||
|
|
||||||
|
if not args.quiet and shown < DISPLAY_LIMIT:
|
||||||
|
print(f"SOPInstanceUID {g['_id']} — {len(docs)} copies indexed:")
|
||||||
|
print(f" KEEP {keep['FilePath']}")
|
||||||
|
for d in remove:
|
||||||
|
stale = "" if os.path.isfile(d["FilePath"]) else " (file no longer on disk)"
|
||||||
|
print(f" REMOVE {d['FilePath']}{stale}")
|
||||||
|
print()
|
||||||
|
shown += 1
|
||||||
|
|
||||||
|
if not args.quiet and len(groups) > DISPLAY_LIMIT:
|
||||||
|
print(f"... and {len(groups) - DISPLAY_LIMIT} more duplicate group(s) not shown.\n")
|
||||||
|
|
||||||
|
print(f"Total index entries that would be removed: {to_remove}")
|
||||||
|
|
||||||
|
if args.dry_run:
|
||||||
|
print("\nRe-run without --dry-run to actually remove these from MongoDB.")
|
||||||
|
return
|
||||||
|
|
||||||
|
if to_remove:
|
||||||
|
result = coll.delete_many({"FilePath": {"$in": paths_to_delete}})
|
||||||
|
deleted = getattr(result, "deleted_count", to_remove)
|
||||||
|
print(f"Removed {deleted} duplicate index entrie(s) from MongoDB.")
|
||||||
|
print()
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Subcommand: recheck-accession
|
||||||
|
#
|
||||||
|
# For files already in the index whose AccessionNumber came back missing,
|
||||||
|
# null, or blank, go back to the actual file on disk and re-read its DICOM
|
||||||
|
# metadata fresh (not from the index). This catches cases like: the
|
||||||
|
# original scan happened before the file was fully written, the file was
|
||||||
|
# re-exported by the modality/PACS with the field filled in afterward, or
|
||||||
|
# the field is genuinely absent from the file itself.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
def _missing_accession_filter():
|
||||||
|
return {
|
||||||
|
"$or": [
|
||||||
|
{"AccessionNumber": {"$exists": False}},
|
||||||
|
{"AccessionNumber": None},
|
||||||
|
{"AccessionNumber": {"$regex": r"^\s*$"}},
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_target_files(coll, files, filenames, note_prefix="recheck"):
|
||||||
|
"""
|
||||||
|
Given raw --file paths and --filename patterns, return a de-duplicated
|
||||||
|
list of absolute file paths to operate on. --filename is resolved by
|
||||||
|
looking up the indexed FileName field (substring, case-insensitive) to
|
||||||
|
find the corresponding FilePath(s); --file paths are used as-is.
|
||||||
|
"""
|
||||||
|
target_files = list(files or [])
|
||||||
|
|
||||||
|
for fn in (filenames or []):
|
||||||
|
matches = list(coll.find({"FileName": {"$regex": fn, "$options": "i"}}, {"_id": 0, "FilePath": 1}))
|
||||||
|
if not matches:
|
||||||
|
print(f"Note: no indexed file found matching filename '{fn}' — skipping.")
|
||||||
|
else:
|
||||||
|
for m in matches:
|
||||||
|
target_files.append(m["FilePath"])
|
||||||
|
|
||||||
|
# De-duplicate while preserving order, in case --file/--filename overlap
|
||||||
|
# or a --filename pattern matched the same file more than once.
|
||||||
|
return [os.path.abspath(f) for f in dict.fromkeys(target_files)]
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Subcommand: inspect
|
||||||
|
#
|
||||||
|
# Read-only: re-reads a file's DICOM metadata straight from disk and prints
|
||||||
|
# it, without touching MongoDB at all (no index writes, no upserts). Useful
|
||||||
|
# for eyeballing what a file actually contains before deciding whether to
|
||||||
|
# rescan/recheck it, or just to sanity-check a specific file/filename.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
def read_dicom_dataset(filepath):
|
||||||
|
"""
|
||||||
|
Read a file with pydicom and return the raw Dataset (not a flattened
|
||||||
|
dict like extract_metadata), or None if it's not a valid DICOM file.
|
||||||
|
Used by --full inspect mode to show every tag actually present.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
return pydicom.dcmread(filepath, stop_before_pixels=True, force=False)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
ds = pydicom.dcmread(filepath, stop_before_pixels=True, force=True)
|
||||||
|
if "SOPClassUID" not in ds and "Modality" not in ds:
|
||||||
|
return None
|
||||||
|
return ds
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _format_all_tags(ds, indent=0):
|
||||||
|
"""Return a list of printable lines covering every element in a Dataset,
|
||||||
|
recursing into sequences (SQ) and truncating/summarizing binary values
|
||||||
|
(pixel data, overlays, etc.) instead of dumping raw bytes."""
|
||||||
|
lines = []
|
||||||
|
pad = " " * indent
|
||||||
|
for elem in ds:
|
||||||
|
keyword = elem.keyword or "(unknown/private tag)"
|
||||||
|
tag_str = str(elem.tag)
|
||||||
|
|
||||||
|
if elem.VR == "SQ":
|
||||||
|
items = elem.value or []
|
||||||
|
lines.append(f"{pad}{tag_str} {keyword:<32} VR=SQ ({len(items)} item(s))")
|
||||||
|
for i, item in enumerate(items, start=1):
|
||||||
|
lines.append(f"{pad} [Item {i}]")
|
||||||
|
lines.extend(_format_all_tags(item, indent + 2))
|
||||||
|
continue
|
||||||
|
|
||||||
|
if elem.VR in ("OB", "OW", "OF", "OL", "OD", "UN") or elem.tag == 0x7FE00010:
|
||||||
|
try:
|
||||||
|
length = len(elem.value) if elem.value is not None else 0
|
||||||
|
except Exception:
|
||||||
|
length = 0
|
||||||
|
lines.append(f"{pad}{tag_str} {keyword:<32} VR={elem.VR:<3} <binary data, {length} bytes>")
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
value_str = str(elem.value)
|
||||||
|
except Exception:
|
||||||
|
value_str = "<unreadable value>"
|
||||||
|
if len(value_str) > 150:
|
||||||
|
value_str = value_str[:150] + "...(truncated)"
|
||||||
|
lines.append(f"{pad}{tag_str} {keyword:<32} VR={elem.VR:<3} {value_str}")
|
||||||
|
|
||||||
|
return lines
|
||||||
|
|
||||||
|
|
||||||
|
def cmd_inspect(args):
|
||||||
|
# Only need MongoDB if resolving --filename patterns to a path; if the
|
||||||
|
# person only passed --file, we never touch the database.
|
||||||
|
coll = None
|
||||||
|
if args.filename:
|
||||||
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
|
target_files = resolve_target_files(coll, args.file, args.filename)
|
||||||
|
if not target_files:
|
||||||
|
print("Nothing to inspect. Pass --file <path> or --filename <name/pattern>.")
|
||||||
|
return
|
||||||
|
|
||||||
|
for filepath in target_files:
|
||||||
|
print(f"\n=== {filepath} ===")
|
||||||
|
if not os.path.isfile(filepath):
|
||||||
|
print(" File not found on disk.")
|
||||||
|
continue
|
||||||
|
|
||||||
|
if args.full:
|
||||||
|
ds = read_dicom_dataset(filepath)
|
||||||
|
if ds is None:
|
||||||
|
print(" Could not be read as a valid DICOM file.")
|
||||||
|
continue
|
||||||
|
for line in _format_all_tags(ds):
|
||||||
|
print(f" {line}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
row = extract_metadata(filepath)
|
||||||
|
if row is None:
|
||||||
|
print(" Could not be read as a valid DICOM file.")
|
||||||
|
continue
|
||||||
|
|
||||||
|
for col_name, _ in DICOM_FIELDS:
|
||||||
|
value = row.get(col_name, "")
|
||||||
|
print(f" {col_name:<22} {value if value != '' else '(empty)'}")
|
||||||
|
print(f" {'FileSizeKB':<22} {row.get('FileSizeKB', '')}")
|
||||||
|
print()
|
||||||
|
|
||||||
|
|
||||||
|
def cmd_recheck_accession(args):
|
||||||
|
coll = get_collection(args.mongo_uri, args.db, args.collection, args.mongo_username, args.mongo_password)
|
||||||
|
|
||||||
|
target_files = resolve_target_files(coll, getattr(args, "file", None), getattr(args, "filename", None))
|
||||||
|
if (getattr(args, "file", None) or getattr(args, "filename", None)) and not target_files:
|
||||||
|
print("Nothing to recheck.")
|
||||||
|
return
|
||||||
|
|
||||||
|
if target_files:
|
||||||
|
# Targeted mode: recheck exactly these file(s), regardless of their
|
||||||
|
# current AccessionNumber status (present, blank, or not indexed at
|
||||||
|
# all yet). Always writes back what's found on disk (an upsert),
|
||||||
|
# since the point is "make sure this specific file is correct in
|
||||||
|
# the index" rather than "only fix missing accession numbers".
|
||||||
|
to_check = [{"FilePath": os.path.abspath(f)} for f in target_files]
|
||||||
|
scope_desc = "specified file(s) targeted for recheck"
|
||||||
|
else:
|
||||||
|
mongo_filter = _missing_accession_filter()
|
||||||
|
if args.patient_id:
|
||||||
|
mongo_filter = {"$and": [mongo_filter, {"PatientID": {"$regex": args.patient_id, "$options": "i"}}]}
|
||||||
|
to_check = list(coll.find(mongo_filter, {"_id": 0, "FilePath": 1}))
|
||||||
|
if args.limit:
|
||||||
|
to_check = to_check[: args.limit]
|
||||||
|
scope_desc = "indexed file(s) currently have a missing/blank AccessionNumber"
|
||||||
|
if args.patient_id:
|
||||||
|
scope_desc += f" for PatientID matching '{args.patient_id}'"
|
||||||
|
|
||||||
|
total = len(to_check)
|
||||||
|
print(f"{total} {scope_desc}.")
|
||||||
|
if total == 0:
|
||||||
|
return
|
||||||
|
|
||||||
|
if args.dry_run:
|
||||||
|
print("(--dry-run given: re-reading files but NOT writing any updates back to MongoDB)\n")
|
||||||
|
print("Re-reading each file's metadata from disk...\n")
|
||||||
|
|
||||||
|
recovered = []
|
||||||
|
still_missing = 0
|
||||||
|
not_found = []
|
||||||
|
read_errors = []
|
||||||
|
batch = []
|
||||||
|
|
||||||
|
for i, doc in enumerate(to_check, start=1):
|
||||||
|
filepath = doc["FilePath"]
|
||||||
|
|
||||||
|
if not os.path.isfile(filepath):
|
||||||
|
not_found.append(filepath)
|
||||||
|
else:
|
||||||
|
row = extract_metadata(filepath)
|
||||||
|
if row is None:
|
||||||
|
read_errors.append(filepath)
|
||||||
|
else:
|
||||||
|
has_accession = bool(row.get("AccessionNumber"))
|
||||||
|
if has_accession:
|
||||||
|
recovered.append((filepath, row["AccessionNumber"]))
|
||||||
|
else:
|
||||||
|
still_missing += 1
|
||||||
|
|
||||||
|
if not args.dry_run:
|
||||||
|
# Targeted-file mode upserts unconditionally (may be a
|
||||||
|
# brand new file not in the index yet). Missing-filter
|
||||||
|
# mode only writes back when something was actually
|
||||||
|
# recovered, since "still missing" files are unchanged.
|
||||||
|
if target_files or has_accession:
|
||||||
|
batch.append(UpdateOne({"FilePath": filepath}, {"$set": row}, upsert=bool(target_files)))
|
||||||
|
|
||||||
|
if not args.quiet and i % 25 == 0:
|
||||||
|
sys.stdout.write(f"\rChecked {i}/{total}...")
|
||||||
|
sys.stdout.flush()
|
||||||
|
|
||||||
|
if len(batch) >= BULK_WRITE_BATCH_SIZE:
|
||||||
|
coll.bulk_write(batch, ordered=False)
|
||||||
|
batch.clear()
|
||||||
|
|
||||||
|
if batch:
|
||||||
|
coll.bulk_write(batch, ordered=False)
|
||||||
|
|
||||||
|
if not args.quiet and total:
|
||||||
|
sys.stdout.write(f"\rChecked {total}/{total}.\n")
|
||||||
|
|
||||||
|
print("\n--- Recheck Results ---")
|
||||||
|
print(f"Files re-checked: {total}")
|
||||||
|
print(f"Accession number recovered: {len(recovered)}" + ("" if args.dry_run else " (updated in MongoDB)"))
|
||||||
|
print(f"Still missing/blank on disk: {still_missing}")
|
||||||
|
print(f"File no longer found on disk: {len(not_found)}")
|
||||||
|
print(f"File unreadable / not DICOM: {len(read_errors)}")
|
||||||
|
|
||||||
|
if recovered:
|
||||||
|
print("\nRecovered accession numbers:")
|
||||||
|
for filepath, acc in recovered[:DISPLAY_LIMIT]:
|
||||||
|
print(f" {acc:<20} {filepath}")
|
||||||
|
if len(recovered) > DISPLAY_LIMIT:
|
||||||
|
print(f" ... and {len(recovered) - DISPLAY_LIMIT} more.")
|
||||||
|
|
||||||
|
if not_found:
|
||||||
|
print("\nFiles missing from disk (index may be stale for these):")
|
||||||
|
for filepath in not_found[:DISPLAY_LIMIT]:
|
||||||
|
print(f" {filepath}")
|
||||||
|
if len(not_found) > DISPLAY_LIMIT:
|
||||||
|
print(f" ... and {len(not_found) - DISPLAY_LIMIT} more.")
|
||||||
|
|
||||||
|
if read_errors:
|
||||||
|
print("\nFiles that could not be read as DICOM on recheck:")
|
||||||
|
for filepath in read_errors[:DISPLAY_LIMIT]:
|
||||||
|
print(f" {filepath}")
|
||||||
|
if len(read_errors) > DISPLAY_LIMIT:
|
||||||
|
print(f" ... and {len(read_errors) - DISPLAY_LIMIT} more.")
|
||||||
|
print()
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Helpers
|
# Helpers
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@ -582,22 +941,28 @@ def _print_table(rows, columns):
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
def _add_mongo_args(p):
|
def _add_mongo_args(p):
|
||||||
p.add_argument("--env-file", default=".env",
|
p.add_argument("--env-file", default=".env",
|
||||||
help="Path to a .env file with MONGO_URI/MONGO_DB/MONGO_COLLECTION (default: .env).")
|
help="Path to a .env file with MONGO_URI/MONGO_DB/MONGO_COLLECTION/MONGO_USERNAME/MONGO_PASSWORD (default: .env).")
|
||||||
p.add_argument("--mongo-uri", default=None,
|
p.add_argument("--mongo-uri", default=None,
|
||||||
help=f"MongoDB connection URI. Overrides .env / $MONGO_URI. (default: {DEFAULT_MONGO_URI})")
|
help=f"MongoDB connection URI. Overrides .env / $MONGO_URI. (default: {DEFAULT_MONGO_URI})")
|
||||||
p.add_argument("--db", default=None,
|
p.add_argument("--db", default=None,
|
||||||
help=f"MongoDB database name. Overrides .env / $MONGO_DB. (default: {DEFAULT_DB_NAME})")
|
help=f"MongoDB database name. Overrides .env / $MONGO_DB. (default: {DEFAULT_DB_NAME})")
|
||||||
p.add_argument("--collection", default=None,
|
p.add_argument("--collection", default=None,
|
||||||
help=f"MongoDB collection name. Overrides .env / $MONGO_COLLECTION. (default: {DEFAULT_COLLECTION_NAME})")
|
help=f"MongoDB collection name. Overrides .env / $MONGO_COLLECTION. (default: {DEFAULT_COLLECTION_NAME})")
|
||||||
|
p.add_argument("--mongo-username", default=None,
|
||||||
|
help="MongoDB username. Overrides .env / $MONGO_USERNAME. Not needed if credentials are already in --mongo-uri.")
|
||||||
|
p.add_argument("--mongo-password", default=None,
|
||||||
|
help="MongoDB password. Overrides .env / $MONGO_PASSWORD. Not needed if credentials are already in --mongo-uri.")
|
||||||
|
|
||||||
|
|
||||||
def resolve_mongo_args(args):
|
def resolve_mongo_args(args):
|
||||||
"""
|
"""
|
||||||
Fill in args.mongo_uri/db/collection from (in priority order):
|
Fill in args.mongo_uri/db/collection/username/password from (in priority
|
||||||
|
order):
|
||||||
1. CLI flags, if explicitly given
|
1. CLI flags, if explicitly given
|
||||||
2. The .env file (args.env_file)
|
2. The .env file (args.env_file)
|
||||||
3. Real environment variables
|
3. Real environment variables
|
||||||
4. Hard-coded defaults
|
4. Hard-coded defaults (uri/db/collection only — username/password
|
||||||
|
default to None, meaning "no separate auth, rely on the URI as-is")
|
||||||
"""
|
"""
|
||||||
load_env_file(args.env_file)
|
load_env_file(args.env_file)
|
||||||
|
|
||||||
@ -607,6 +972,10 @@ def resolve_mongo_args(args):
|
|||||||
args.db = os.environ.get("MONGO_DB", DEFAULT_DB_NAME)
|
args.db = os.environ.get("MONGO_DB", DEFAULT_DB_NAME)
|
||||||
if args.collection is None:
|
if args.collection is None:
|
||||||
args.collection = os.environ.get("MONGO_COLLECTION", DEFAULT_COLLECTION_NAME)
|
args.collection = os.environ.get("MONGO_COLLECTION", DEFAULT_COLLECTION_NAME)
|
||||||
|
if args.mongo_username is None:
|
||||||
|
args.mongo_username = os.environ.get("MONGO_USERNAME") or None
|
||||||
|
if args.mongo_password is None:
|
||||||
|
args.mongo_password = os.environ.get("MONGO_PASSWORD") or None
|
||||||
|
|
||||||
|
|
||||||
def build_parser():
|
def build_parser():
|
||||||
@ -648,6 +1017,10 @@ def build_parser():
|
|||||||
p_search.add_argument("--series", help="Filter by series description (substring match).")
|
p_search.add_argument("--series", help="Filter by series description (substring match).")
|
||||||
p_search.add_argument("--body-part", dest="body_part", help="Filter by body part examined.")
|
p_search.add_argument("--body-part", dest="body_part", help="Filter by body part examined.")
|
||||||
p_search.add_argument("--accession", help="Filter by accession number.")
|
p_search.add_argument("--accession", help="Filter by accession number.")
|
||||||
|
p_search.add_argument(
|
||||||
|
"--accession-missing", action="store_true",
|
||||||
|
help="Only show records where AccessionNumber is missing, null, or blank (overrides --accession).",
|
||||||
|
)
|
||||||
p_search.add_argument("--columns", help="Comma-separated list of columns to display (default: a sensible subset).")
|
p_search.add_argument("--columns", help="Comma-separated list of columns to display (default: a sensible subset).")
|
||||||
p_search.add_argument("-o", "--output", help="Save filtered results to this CSV path.")
|
p_search.add_argument("-o", "--output", help="Save filtered results to this CSV path.")
|
||||||
p_search.set_defaults(func=cmd_search)
|
p_search.set_defaults(func=cmd_search)
|
||||||
@ -663,6 +1036,85 @@ def build_parser():
|
|||||||
p_export.add_argument("-o", "--output", default="dicom_index.csv", help="Output CSV path (default: dicom_index.csv).")
|
p_export.add_argument("-o", "--output", default="dicom_index.csv", help="Output CSV path (default: dicom_index.csv).")
|
||||||
p_export.set_defaults(func=cmd_export)
|
p_export.set_defaults(func=cmd_export)
|
||||||
|
|
||||||
|
# dedupe
|
||||||
|
p_dedupe = subparsers.add_parser(
|
||||||
|
"dedupe",
|
||||||
|
help="Find and remove duplicate index entries — the same DICOM instance indexed under more than one file path.",
|
||||||
|
)
|
||||||
|
_add_mongo_args(p_dedupe)
|
||||||
|
p_dedupe.add_argument(
|
||||||
|
"--dry-run", action="store_true",
|
||||||
|
help="Report duplicates found without deleting anything from MongoDB.",
|
||||||
|
)
|
||||||
|
p_dedupe.add_argument(
|
||||||
|
"--keep", choices=["newest", "oldest"], default="newest",
|
||||||
|
help="Which copy to keep per duplicate group, by IndexedAt (default: newest). "
|
||||||
|
"A copy whose file still exists on disk is always preferred over one that's gone missing.",
|
||||||
|
)
|
||||||
|
p_dedupe.add_argument("-q", "--quiet", action="store_true", help="Suppress per-group detail, show totals only.")
|
||||||
|
p_dedupe.set_defaults(func=cmd_dedupe)
|
||||||
|
|
||||||
|
# recheck-accession
|
||||||
|
p_recheck = subparsers.add_parser(
|
||||||
|
"recheck-accession",
|
||||||
|
help="Re-read on-disk DICOM metadata for indexed files whose AccessionNumber is missing/blank.",
|
||||||
|
)
|
||||||
|
_add_mongo_args(p_recheck)
|
||||||
|
p_recheck.add_argument(
|
||||||
|
"--dry-run", action="store_true",
|
||||||
|
help="Re-read files and report findings only; don't write any updates back to MongoDB.",
|
||||||
|
)
|
||||||
|
p_recheck.add_argument(
|
||||||
|
"--limit", type=int, default=None,
|
||||||
|
help="Only check the first N matching files (useful to sample a large index before a full run).",
|
||||||
|
)
|
||||||
|
p_recheck.add_argument(
|
||||||
|
"--patient-id", dest="patient_id", default=None,
|
||||||
|
help="Only recheck missing-accession files for this PatientID (substring match).",
|
||||||
|
)
|
||||||
|
p_recheck.add_argument(
|
||||||
|
"--file", action="append", default=None,
|
||||||
|
help=(
|
||||||
|
"Recheck one specific file path instead of scanning the whole index for "
|
||||||
|
"missing accession numbers. Can be given multiple times to check several "
|
||||||
|
"files. Works even if the file isn't in the index yet."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
p_recheck.add_argument(
|
||||||
|
"--filename", action="append", default=None,
|
||||||
|
help=(
|
||||||
|
"Recheck file(s) by filename (substring match against the indexed FileName) "
|
||||||
|
"instead of a full path — handy when you know the filename but not its full "
|
||||||
|
"location. Can be given multiple times. Only matches files already in the index."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
p_recheck.add_argument("-q", "--quiet", action="store_true", help="Suppress progress output.")
|
||||||
|
p_recheck.set_defaults(func=cmd_recheck_accession)
|
||||||
|
|
||||||
|
# inspect
|
||||||
|
p_inspect = subparsers.add_parser(
|
||||||
|
"inspect",
|
||||||
|
help="Read-only: show all DICOM metadata for specific file(s), straight from disk. No MongoDB writes.",
|
||||||
|
)
|
||||||
|
_add_mongo_args(p_inspect)
|
||||||
|
p_inspect.add_argument(
|
||||||
|
"--file", action="append", default=None,
|
||||||
|
help="Full path to a file to inspect. Can be given multiple times. Doesn't need to be indexed.",
|
||||||
|
)
|
||||||
|
p_inspect.add_argument(
|
||||||
|
"--filename", action="append", default=None,
|
||||||
|
help="Inspect file(s) by filename (substring match against the indexed FileName). Can be given multiple times.",
|
||||||
|
)
|
||||||
|
p_inspect.add_argument(
|
||||||
|
"--full", action="store_true",
|
||||||
|
help=(
|
||||||
|
"Dump EVERY DICOM tag found in the file, not just the fields normally "
|
||||||
|
"indexed. Useful for locating unusual/site-specific fields (e.g. a "
|
||||||
|
"'registration number' stored under OtherPatientIDs or a private tag)."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
p_inspect.set_defaults(func=cmd_inspect)
|
||||||
|
|
||||||
return parser
|
return parser
|
||||||
|
|
||||||
|
|
||||||
@ -674,4 +1126,4 @@ def main():
|
|||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
Loading…
x
Reference in New Issue
Block a user