Repository navigation
Expand file tree
/
Copy pathrimport
More file actions
executable file
·920 lines (761 loc) · 39.5 KB
/
Copy pathrimport
File metadata and controls
executable file
·920 lines (761 loc) · 39.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
#!/glade/u/apps/derecho/24.12/opt/view/bin/python
# TODO: Move all the Python into new file rimport.py for simpler testing. Keep rimport as a
# convenience wrapper.
"""
Copy files from CESM inputdata directory to a publishing directory, then replace the original with a
symlink to the copy.
Do `rimport --help` for more information.
"""
from __future__ import annotations
import argparse
import os
import pwd
import shutil
import sys
from pathlib import Path
from typing import Iterable, List, NamedTuple
from urllib.request import Request, urlopen
from urllib.error import HTTPError
from relink import replace_one_file_with_symlink
import shared
INDENT = shared.INDENT
DEFAULT_INPUTDATA_ROOT = Path(shared.DEFAULT_INPUTDATA_ROOT)
DEFAULT_STAGING_ROOT = Path(shared.DEFAULT_STAGING_ROOT)
STAGE_OWNER = "cesmdata"
INPUTDATA_URL = "https://osdf-data.gdex.ucar.edu/ncar/gdex/d651077/cesmdata/inputdata"
# Configure logging
logger = shared.logger
class _EpilogRawFormatter(argparse.HelpFormatter):
"""Wrap the description as argparse normally would, but leave the epilog verbatim.
RawDescriptionHelpFormatter would preserve the epilog's exit-code list, but it also
stops re-wrapping the description, which then renders as one long line on a narrow
terminal. Only the epilog needs its line breaks kept.
"""
def _fill_text(self, text, width, indent):
if text.lstrip().startswith("exit codes:"):
return "".join(indent + line for line in text.splitlines(keepends=True))
return super()._fill_text(text, width, indent)
def build_parser() -> argparse.ArgumentParser:
"""Build and configure the argument parser for rimport.
Returns:
argparse.ArgumentParser: Configured parser ready to parse command-line arguments.
"""
parser = argparse.ArgumentParser(
description=(
f"Copy files from CESM inputdata directory ({DEFAULT_INPUTDATA_ROOT}) to a publishing"
" directory, then replace the original with a symlink to the copy."
),
epilog=(
"exit codes:\n"
" 0: everything staged or checked, nothing skipped\n"
" 1: a file could not be staged\n"
" 2: nothing was published -- a bad command line, a missing or empty\n"
" --list file, or a name you gave failing validation\n"
" 3: finished, but one or more items were skipped (listed at the end)\n"
"\n"
"a name you gave failing is fatal (2). anything found by expanding a\n"
"directory you named -- a bad file, or a subdirectory that cannot be\n"
"read -- is skipped instead, and the run continues (3).\n"
"when several apply, the precedence is 2 > 1 > 3 > 0.\n"
),
formatter_class=_EpilogRawFormatter,
add_help=False, # Disable automatic help to add custom -help flag
)
parser.add_argument(
"--file",
"-file",
dest="file",
metavar="filename",
help=(
"Provide a file to import. Must be in the CESM inputdata directory. A relative name"
" is resolved against the current directory; there is no fallback to the"
" inputdata root. If the name is a directory, every file beneath it is"
" enumerated recursively and acted on; a symlink to a directory is not expanded."
),
)
parser.add_argument(
"--list",
"-list",
dest="filelist",
metavar="filelist",
help=(
"Provide a file that contains a list of filenames to import. All filenames in the list"
" must be in the CESM inputdata directory. A relative entry is resolved against the"
" list file's own directory, wherever that directory is. A list entry naming a"
" directory is enumerated recursively, the same as a name given on the command"
" line."
),
)
parser.add_argument(
"items_to_process",
nargs="*",
help=(
"One or more files to process. (Optional; can use --file instead to process just one.)"
" Must be in the CESM inputdata directory. A relative name is resolved against the"
" current directory; there is no fallback to the inputdata root. If the name is a"
" directory, every file beneath it is enumerated recursively and acted on; a symlink"
" to a directory is not expanded."
),
)
# Add inputdata_root option flags
shared.add_inputdata_root(parser)
parser.add_argument(
"--check",
"-check",
"-c",
action="store_true",
help=(
"Check whether item(s) is/are already published, without staging anything. A bad"
" name that you gave aborts before any file is checked, reporting all bad names"
" at once; anything found by enumerating a directory is instead reported and"
" skipped individually."
),
)
# Add verbosity options
shared.add_parser_verbosity_group(parser)
# Add help text
shared.add_help(parser)
return parser
def read_filelist(list_path: Path) -> List[str]:
"""Read a file list and return non-empty, non-comment lines.
Reads a text file containing a list of filenames, filtering out:
- Blank lines (empty or whitespace-only)
- Comment lines (starting with '#')
Args:
list_path: Path to the file containing the list of filenames.
Returns:
List of stripped, non-empty lines that are not comments.
"""
lines: List[str] = []
with list_path.open("r", encoding="utf-8") as f:
for raw in f:
line = raw.strip()
if not line or line.startswith("#"):
continue
lines.append(line)
return lines
def normalize_paths(root: Path, relnames: Iterable[str]) -> List[Path]:
"""Convert relative or absolute path names to normalized absolute Paths.
For each name in relnames:
- If the name is relative, it is joined onto `root` and made absolute.
All paths are then normalized to their absolute form, replacing . and .. as needed.
The `root / name` branch above is unreachable from `main`: `get_files_to_process` now
anchors every relative name before returning it -- CLI-style names (`--file`, positional)
against cwd, `--list` entries against the list file's own directory -- so nothing relative
ever reaches this function on `main`'s call path, and `root` is effectively unused there.
This is dead code, kept deliberately rather than removed; do not read this function in
isolation and conclude that root-relative resolution is still reachable anywhere in
`rimport`. The branch is still real code, though: it is exercised directly by this
function's own unit tests in test_normalize_paths.py, which call normalize_paths() with
relative names and a `root` of their choosing.
Note that symlinks are NOT resolved.
Args:
root: Base directory under which relative paths are assumed to be. Only matters for
the unreachable-from-`main` branch described above.
relnames: Iterable of path names (relative or absolute) to normalize.
Returns:
List of normalized absolute Path objects.
"""
paths: List[Path] = []
for name in relnames:
p = root / name if not Path(name).is_absolute() else Path(name)
p = Path(os.path.normpath(p.absolute()))
paths.append(p)
return paths
class Entry(NamedTuple):
"""One path to consider staging, and how it got into the batch.
`named` is True when the user typed this path (positional, `--file`, or a `--list`
entry) and False when a directory walk discovered it. The distinction is what lets a
bad file inside a large tree warn-and-skip while a bad path the user asked for by name
still aborts the whole batch.
"""
path: Path
named: bool
class Skip(NamedTuple):
"""One path that will not be staged, the reason why, and whether the user named it.
Both validation failures on discovered files and directories that could not be read
become Skips, so `report_skips` can present them in one block.
`named` defaults to False so a walker that has no idea what the user typed can build a
Skip with two arguments. `expand_directories` is the one place that knows the whole
batch, so it is the one place that fills this in; `main` then trusts it rather than
re-deriving it, which previously meant two copies of the same set having to agree.
"""
path: Path
reason: Exception
named: bool = False
def walk_files(root: Path) -> tuple[List[Path], List[Skip]]:
"""Recursively list the non-directory entries under `root`.
Descends only into real directories (`follow_symlinks=False`), so the walk cannot
leave the tree the user named and cannot loop on a cyclic link. A symlink is therefore
always a leaf: it is returned as an entry to act on, including when its target happens
to be a directory, and `validate_source_path` decides what that means.
Every non-directory entry is returned -- regular files, symlinks, dotfiles alike. There
is no owner filter (unlike relink's walker): rimport runs as the staging owner and the
point is to publish what is there.
Entries are sorted at each level so output and tests are deterministic; `os.scandir`
order is otherwise arbitrary.
Args:
root: Directory to walk.
Returns:
(files, skips). `files` are absolute paths in depth-first, per-level sorted order.
`skips` are directories that could not be read; an unreadable directory does not
raise and does not abort the rest of the walk.
"""
files: List[Path] = []
skips: List[Skip] = []
try:
with os.scandir(root) as scan:
children = sorted(scan, key=lambda entry: entry.name)
except OSError as exc:
return [], [Skip(Path(root), exc)]
for child in children:
child_path = Path(child.path)
if child.is_dir(follow_symlinks=False):
sub_files, sub_skips = walk_files(child_path)
files.extend(sub_files)
skips.extend(sub_skips)
else:
files.append(child_path)
return files, skips
def expand_directories(
paths: Iterable[Path], inputdata_root: Path
) -> tuple[List[Entry], List[Skip]]:
"""Replace each directory in `paths` with the files beneath it, tagging provenance.
A path is expanded only if it is a real directory INSIDE `inputdata_root`. A symlink to
a directory is left alone as a single named entry, matching `walk_files`' refusal to
descend through one. Nothing here validates: a nonexistent path, or a directory outside
the tree, passes straight through as named, so `main`'s pre-flight gate can reject it
with its usual message. Declining to expand is not a verdict -- it just leaves the path
for the gate that already knows how to judge it.
Scoping expansion to the tree matters for more than tidiness. Walking recurses, so
expanding a directory this tool has no business in -- a mistyped `rimport ~` -- would
stat an arbitrarily large tree before rejecting every file in it, and would downgrade
the user's own bad argument from a fatal named failure to a heap of discovered skips.
Duplicates collapse in first-seen order. If the same path is both named and discovered
-- the user typed a file that also lives inside a directory they named -- `named` wins,
so the path the user asked for by name keeps its fatal-on-failure treatment.
Args:
paths: Absolute paths from `normalize_paths`.
inputdata_root: Root of the inputdata tree; only directories under it are expanded.
Returns:
(entries, skips), each skip carrying its own provenance. `skips` holds two things:
directories that could not be read during a walk, and named paths whose is_dir()
probe itself raised (a file under an unreadable parent, say). A walk skip is usually
discovered, and a discovered skip warns and continues with exit 3 rather than
aborting; only a skip the user named is fatal. Skips are collapsed by path, so a
directory named twice yields one.
"""
# Consumed twice -- once to walk, once to test provenance -- so a generator will not do.
# Collapsing here means a directory named twice is walked once, so it counts once toward
# the expansion total and reports anything unreadable beneath it once.
paths = list(dict.fromkeys(paths))
named_paths = set(paths)
named_by_path: dict[Path, bool] = {}
skips: List[Skip] = []
empty: List[Path] = []
n_dirs = 0
expanded_files: set[Path] = set()
for path in paths:
try:
is_expandable_dir = path.is_dir() and not path.is_symlink()
except OSError as exc:
# Path.is_dir() ignores only ENOENT, ENOTDIR, EBADF and ELOOP. It propagates
# everything else -- EACCES included, on every supported version -- so a path
# under an unreadable parent would abort the run with a traceback. Record it and
# let main decide: every path here was NAMED by the user, so main routes it to
# the fatal pre-flight block and the run exits 2 having published nothing.
skips.append(Skip(path, exc))
continue
if is_expandable_dir and not path.resolve().is_relative_to(inputdata_root.resolve()):
# Outside the tree: hand it to the pre-flight gate unexpanded. Resolve both
# sides, as validate_source_path does, so the two agree about what "outside"
# means for a path reached through a symlinked parent.
is_expandable_dir = False
if is_expandable_dir:
n_dirs += 1
found, walk_skips = walk_files(path)
skips.extend(walk_skips)
if not found and not walk_skips:
# Collected, not reported: every warning this function emits is logged
# below the count. A walk that returned nothing because it could not be
# read is not an empty directory, and is reported as a skip instead.
empty.append(path)
for found_path in found:
expanded_files.add(found_path)
named_by_path.setdefault(found_path, False)
else:
# A named path always wins over the same path discovered by a walk.
named_by_path[path] = True
# Collapse by path and stamp provenance before anything counts or prints these. Two
# named directories can overlap -- `<tree>` and `<tree>/sub` -- and then the same
# unreadable grandchild is found by both walks.
deduped: dict[Path, Skip] = {}
for skip in skips:
deduped.setdefault(skip.path, Skip(skip.path, skip.reason, skip.path in named_paths))
skips = list(deduped.values())
# Everything below is logged after the count, and indented under it. The count is the
# blast radius -- the one line worth reading before a large run is left to finish -- so
# it stays at the top of the output however many directories turn out to warn.
if n_dirs:
logger.info(
"rimport: expanded %d director(ies) to %d file(s)", n_dirs, len(expanded_files)
)
for path in empty:
logger.warning("%srimport: no files found under %s", INDENT, path)
for skip in skips:
# Report a DISCOVERED skip here, during expansion, where the run reached it; the
# end-of-run summary repeats it. A skip the user NAMED is not reported here at all:
# main treats it as a fatal failure, and calling it "skipping" would contradict the
# "nothing was published" that follows. Note this asks about the whole batch, not
# about the directory being walked: a directory can be named AND discovered beneath
# another named one, and then both are true of it.
if not skip.named:
logger.warning(
"%srimport: skipping '%s': %s", INDENT, skip.path, reason_text(skip.reason)
)
entries = [Entry(path, named) for path, named in named_by_path.items()]
return entries, skips
def report_skips(skips: List[Skip]) -> None:
"""Re-list every skipped path at the very end of the run.
Each skip was already reported inline, at WARNING level, where it happened. This block
repeats them at ERROR level -- so stderr, and so surviving `-q` -- because in a run over
a large tree the inline warnings scroll away and the whole point is that a file which
went unpublished cannot be missed.
Args:
skips: Every skipped path with its reason. An empty list prints nothing.
"""
if not skips:
return
logger.error("rimport: %d item(s) skipped (not stageable):", len(skips))
for skip in skips:
logger.error("%s%s: %s", INDENT, skip.path, reason_text(skip.reason))
def reason_text(reason: Exception) -> str:
"""Render a skip or failure reason for a message that already names the path.
`OSError`'s str() appends the filename, so a line built as "'<path>': <reason>" prints
the path twice. Its `strerror` is the same message without that, and errno is dropped
deliberately: "Permission denied" is what the user can act on. This is safe only because
every OSError reaching here was raised on the same path the message names.
It changes nothing for the RuntimeErrors `validate_source_path` builds: those carry no
`strerror`, and several write the path into their own text, which this cannot undo.
"""
if isinstance(reason, OSError) and reason.strerror:
return reason.strerror
return str(reason)
def check_relink_worked(src: Path, dst: Path) -> None:
"""Check whether relink worked
Args:
src (Path): Source file (should have been converted to symlink)
dst (Path): Destination file (symlink target)
Raises:
RuntimeError: If src is not a symlink pointing to dst.
"""
if not (src.is_symlink() and src.resolve() == dst):
raise RuntimeError("Error relinking during rimport")
def validate_source_path(
src: Path, inputdata_root: Path, staging_root: Path
) -> Exception | None:
"""Run stage_data's read-only guardrails against `src` without raising or logging.
This is the pre-flight half of `stage_data`'s checks, in the order they run: broken
symlink, live symlink whose target is outside staging, missing file, outside the inputdata
root, already under the staging directory, and directory. It performs no I/O beyond
stat-ing `src` and its resolved target — in particular it never makes a network call — so
it is cheap to run over an entire batch before anything is staged.
Critically, a *live* symlink whose target resolves under `staging_root` is NOT a failure:
that is the normal state of a file that has already been published and linked by a previous
run, and is treated here exactly like a plain, stageable file (returns `None`). Re-running
rimport over an already-published tree must not report every published file as bad.
`stage_data` calls this first and raises whatever it returns; `main`'s pre-flight gate calls
it over every resolved path before staging anything, so a bad path anywhere in the batch is
reported without partially publishing the rest.
Args:
src: Source file path to validate.
inputdata_root: Root directory of the inputdata tree.
staging_root: Root directory where files will be staged.
Returns:
The exception (unraised) describing why `src` cannot be staged, or `None` if `src` is
fine to proceed with — including the already-published-and-linked symlink case.
"""
if src.is_symlink():
if not os.path.exists(src.resolve()):
return RuntimeError(f"Source is a broken symlink: {src}")
if not src.resolve().is_relative_to(staging_root.resolve()):
return RuntimeError(
f"Source is a symlink, but target ({src.resolve()}) is outside staging directory "
f"({staging_root})"
)
# Live symlink already resolving under staging_root: already published and linked.
# This is a legitimate no-op, not a validation failure.
return None
if not src.exists():
return FileNotFoundError(f"source not found: {src}")
# Containment is checked before the is-a-directory backstop so that a DIRECTORY outside
# the tree names the reason the user can act on. Told "source is a directory, not a
# file", they would reasonably reply that directories are supported now.
try:
src.resolve().relative_to(inputdata_root.resolve())
except ValueError:
if src.resolve().is_relative_to(staging_root.resolve()):
return RuntimeError(
f"Source file '{src.name}' is already under staging directory '{staging_root}'."
)
return RuntimeError(
f"source not under inputdata root: {src} not in {inputdata_root}"
)
if src.is_dir():
return RuntimeError(f"source is a directory, not a file: {src}")
return None
def stage_data(
src: Path, inputdata_root: Path, staging_root: Path, check: bool = False
) -> None:
"""Stage a file by mirroring its path under `staging_root`, then replace with symlink to staged.
Destination path is computed by replacing the `inputdata_root` prefix of `src`
with `staging_root`, i.e.:
dst = staging_root / src.relative_to(inputdata_root)
Args:
src: Source file path to stage.
inputdata_root: Root directory of the inputdata tree.
staging_root: Root directory where files will be staged.
check: If True, just check whether the file is already published.
Raises:
RuntimeError: If `src` is a live symlink pointing outside staging, or if `src` is outside
the inputdata root, or if `src` is already under staging directory.
RuntimeError: If `src` is a broken symlink.
RuntimeError: If `src` is a directory. This check runs only when `src` is not itself
a symlink — see the symlink guardrails above, which return early (without
raising) for a symlink whose target is a directory under `staging_root`.
A symlink whose directory target is outside `staging_root` instead raises
via the first entry above.
RuntimeError: If it failed to replace `src` with a symlink to the staged file.
FileNotFoundError: If `src` does not exist.
Guardrails:
* Raise if `src` is a *live* symlink to a target outside staging root ("outside staging").
* Raise if `src` is a broken symlink or is outside the inputdata root.
* Raise if `src` is a directory and not itself a symlink, so a non-symlink directory
source can never reach the replace-with-symlink path (which assumes a regular file
and mangles a directory). A *symlink* whose target is a directory is NOT covered by
this guardrail: it is handled by the live-symlink guardrails above instead, which for
a target under `staging_root` log "already published and linked" and return without
raising, and for a target outside `staging_root` raise (see the first guardrail above).
These read-only guardrails are delegated to `validate_source_path`, which returns the
exception to raise (or None); this function raises whatever comes back. That keeps this
function and `main`'s pre-flight gate in permanent agreement about what is stageable,
since both call the same check.
"""
error = validate_source_path(src, inputdata_root, staging_root)
if error is not None:
raise error
if src.is_symlink():
# validate_source_path only lets a symlink through here when it is live and its target
# resolves under staging_root: the "already published and linked" case.
logger.info("%sFile is already published and linked.", INDENT)
print_can_file_be_downloaded(
can_file_be_downloaded(src.resolve(), staging_root)
)
return
rel = src.resolve().relative_to(inputdata_root.resolve())
dst = staging_root / rel
if dst.exists():
msg = "File is already published but NOT linked"
if check:
logger.info("%s; would link.", msg)
print_can_file_be_downloaded(can_file_be_downloaded(rel, staging_root))
else:
logger.info("%s; linking now.", msg)
replace_one_file_with_symlink(inputdata_root, staging_root, str(src))
check_relink_worked(src, dst)
print_can_file_be_downloaded(can_file_be_downloaded(rel, staging_root))
return
if check:
logger.info("%sFile is not already published", INDENT)
return
# Copy file to destination
dst.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(src, dst)
logger.info("%s[rimport] staged %s -> %s", INDENT, src, dst)
# Replace original with symlink to destination
replace_one_file_with_symlink(inputdata_root, staging_root, str(src))
check_relink_worked(src, dst)
def ensure_running_as(target_user: str, argv: list[str]) -> None:
"""Ensure the script is running as the target user, re-executing via sudo if needed.
If not running as `target_user`, re-exec via sudo -u target_user (handles 2FA via PAM).
This function will not return if re-execution is needed; it replaces the current process.
Args:
target_user: Username to run as (e.g., 'cesmdata').
argv: Command-line arguments to pass to the re-executed process.
Raises:
SystemExit: If the target user is not found on the system (exit code 2).
SystemExit: If not running interactively and authentication is required (exit code 2).
Note:
If re-execution is needed, this function calls os.execvp() and does not return.
"""
try:
target_uid = pwd.getpwnam(target_user).pw_uid
except KeyError as exc:
logger.error("rimport: target user '%s' not found on this system", target_user)
raise SystemExit(2) from exc
if os.geteuid() != target_uid:
try:
assert sys.stdin.isatty()
except AssertionError as exc:
logger.error(
"rimport: need interactive TTY to authenticate as '%s' (2FA).\n"
" Try: sudo -u %s rimport …",
target_user,
target_user,
)
raise SystemExit(2) from exc
# Re-exec under target user; this invokes sudo’s normal password/2FA flow.
os.execvp("sudo", ["sudo", "-u", target_user, "--"] + argv)
def get_staging_root() -> Path:
"""Return the staging root directory path.
Uses $RIMPORT_STAGING if set, otherwise returns the default staging root.
Returns:
Path: Resolved absolute path to the staging root directory.
"""
env = os.getenv("RIMPORT_STAGING")
if env:
return Path(env).expanduser().resolve()
return DEFAULT_STAGING_ROOT
def can_file_be_downloaded(file_relpath: Path, staging_root: Path, timeout: float = 10):
"""Check whether a file is available for download from the CESM inputdata server.
Sends a HEAD request to the CESM inputdata URL to verify if the file exists and is
accessible without downloading the entire file.
Args:
file_relpath: Relative path to the file (relative to staging_root), or an absolute
path that will be made relative to staging_root.
staging_root: Root directory of the staging area, used to compute relative path
if file_relpath is absolute.
timeout: Maximum time in seconds to wait for the server response. Default is 10.
Returns:
bool: True if the file is accessible (HTTP status 2xx or 3xx), False otherwise
(including 404, network errors, timeouts, etc.).
"""
# Get URL
if file_relpath.is_absolute():
file_relpath = file_relpath.relative_to(staging_root)
url = os.path.join(INPUTDATA_URL, file_relpath)
# Check whether URL can be accessed
req = Request(url, method="HEAD")
try:
with urlopen(req, timeout=timeout) as resp:
return 200 <= resp.status < 400
except HTTPError:
# Server reached, but resource doesn't exist (404, 410, etc.)
return False
def print_can_file_be_downloaded(file_can_be_downloaded: bool):
"""Print a message indicating whether a file is available for download.
Args:
file_can_be_downloaded: Boolean indicating if the file can be downloaded.
"""
if file_can_be_downloaded:
logger.info("%sFile is available for download.", INDENT)
else:
logger.info("%sFile is not (yet) available for download.", INDENT)
def get_files_to_process(file: str, filelist: str, items_to_process: list):
"""Get list of files to process.
Uses --file and/or --filelist arguments, as well as positional items_to_process if given.
One rule, no modes, no root-relative fallback: non-absolute `file` and
`items_to_process` entries are CLI args and always anchor (eagerly, absolute) to cwd.
Non-absolute `--list` entries always anchor (eagerly, absolute) against the list
file's own directory. Absolute paths are returned unchanged everywhere.
If `Path.cwd()` itself raises (e.g. the working directory was deleted out from under
the process) and at least one `file`/`items_to_process` entry is relative, there is
nothing to anchor it against: this is fatal, reporting every offending relative name
in one error message and returning `(None, 2)`. Absolute `file`/`items_to_process`
entries, and all `--list` entries (which never anchor to cwd), are unaffected by an
undeterminable cwd.
An empty or whitespace-only `file`/`items_to_process` entry is rejected with `(None, 2)`
before any anchoring happens, because an empty name would otherwise resolve to the cwd
and be expanded.
Args:
file (str): Single file to process.
filelist (str): File containing list of files to process.
items_to_process (list): List of files to process.
Returns:
list: List of files to process
int: Result code
"""
cli_args = ([file] if file is not None else []) + list(items_to_process or [])
# An empty name anchors to cwd and resolves to the cwd itself, which then expands: an
# unset shell variable (`rimport "$maybe_unset"`) would recursively publish the whole
# subtree the user is standing in. Refuse it.
empty_cli_args = [name for name in cli_args if not str(name).strip()]
if empty_cli_args:
logger.error(
"rimport: %d empty filename argument(s) given; refusing to resolve an empty "
"name to the current directory. Did a shell variable expand to nothing?",
len(empty_cli_args),
)
return None, 2
relative_cli_args = [name for name in cli_args if not Path(name).is_absolute()]
cwd = None
if relative_cli_args:
try:
cwd = Path.cwd().resolve()
except OSError:
# cwd may have been deleted out from under the process (FileNotFoundError, itself
# an OSError). There is no root fallback to anchor a relative name against instead,
# so this is fatal: report every offending name in one error message.
logger.error(
"rimport: could not determine the current working directory (it may have "
"been deleted); cannot resolve relative name(s): %s",
", ".join(relative_cli_args),
)
return None, 2
def _anchor_cli(name):
if Path(name).is_absolute():
return name
return str(cwd / name)
files_to_process = [_anchor_cli(file)] if file is not None else []
if filelist is not None:
list_path = Path(filelist).expanduser().resolve()
if not list_path.exists():
logger.error("rimport: list file not found: %s", list_path)
return None, 2
files_in_list = read_filelist(list_path)
if not files_in_list:
logger.error("rimport: no filenames found in list: %s", list_path)
return None, 2
list_base = list_path.parent
for entry in files_in_list:
if Path(entry).is_absolute():
files_to_process.append(entry)
else:
files_to_process.append(str(list_base / entry))
if items_to_process:
files_to_process.extend(_anchor_cli(item) for item in items_to_process)
if not files_to_process:
logger.error("rimport: At least one of --file or --filelist is required")
return None, 2
return files_to_process, 0
def main(argv: List[str] | None = None) -> int:
"""Main entry point for the rimport tool.
Copies and relinks files from the CESM inputdata directory to a staging/publishing directory,
preserving the directory structure. Ensures the script runs as the correct user
(STAGE_OWNER) and handles both single files and file lists.
Args:
argv: Command-line arguments to parse. If None, uses sys.argv.
Returns:
int: Exit code (0 for success with nothing skipped, 1 if any files had errors,
2 for fatal errors, or 3 if the run completed but one or more paths were
skipped).
Environment Variables:
RIMPORT_SKIP_USER_CHECK: Set to "1" to skip automatic user switching.
RIMPORT_STAGING: Override the default staging root directory.
Exit Codes:
0: All files staged successfully (or, under --check, checked without error), and
nothing was skipped.
1: Pre-flight validation passed, but one or more files failed while actually being
staged or relinked -- a genuine runtime failure, not a rejected input (errors
printed to stderr for each).
2: The run was rejected before any file was staged or checked. Causes:
* argparse rejected the command line itself -- e.g. an unrecognized option,
or an --inputdata-root that does not exist (argparse validates it via
type=, so this fires before any of the code below runs).
* The inputdata root directory does not exist.
* No --file, --list, or positional argument was given.
* The --list file was not found, or contained no filenames.
* A relative --file or positional name could not be anchored, because the
current working directory could not be determined.
* User-switching to STAGE_OWNER failed: target user not found, or no TTY
available for the sudo/2FA prompt.
* Most commonly: pre-flight validation rejected one or more of the resolved
paths (missing file, directory, broken symlink, source outside the
inputdata root, etc.). Pre-flight checks every resolved path before
staging anything and reports every failure at once, so a bad path never
leaves the batch half-published. This gate applies to --check runs too.
3: The run finished -- nothing was rejected outright and no file failed while being
staged or relinked -- but at least one discovered path was skipped rather than
staged. Two things cause that, and either alone is enough: a file found by
expanding a directory argument failed pre-flight validation, or a directory
beneath a named one could not be read. Each skip is warned about inline during
expansion or pre-flight and the full list is repeated on stderr at the very end,
so a skipped path cannot be missed. Note that only DISCOVERED paths are skipped;
the same failure on a path the user named is fatal, and exits 2.
"""
parser = build_parser()
args = parser.parse_args(argv)
# Configure logging based on verbosity flags
log_level = shared.get_log_level(quiet=args.quiet, verbose=args.verbose)
shared.configure_logging(log_level)
# Ensure we are running as the STAGE_OWNER account before touching the tree
# Set env var RIMPORT_SKIP_USER_CHECK=1 if you prefer to run `sudox -u STAGE_OWNER rimport …`
# explicitly (or for testing).
if not args.check and os.getenv("RIMPORT_SKIP_USER_CHECK") != "1":
ensure_running_as(STAGE_OWNER, sys.argv)
root = Path(args.inputdata_root).expanduser().resolve()
if not root.exists():
logger.error("rimport: inputdata directory does not exist: %s", root)
return 2
# Determine the list of relative filenames to handle
files_to_process, status = get_files_to_process(
args.file, args.filelist, args.items_to_process
)
if status:
return status
# Resolve to full paths (keep accepting absolute names too)
paths = normalize_paths(root, files_to_process)
staging_root = get_staging_root()
# Expand any directory argument into the files beneath it, tagging each path as named
# (the user asked for it) or discovered (a walk found it).
entries, skips = expand_directories(paths, root)
# Pre-flight: validate everything before staging anything. A NAMED bad path is fatal and
# aborts the whole batch, so a batch containing one is never half-completed. A DISCOVERED
# bad path is only a skip: one unstageable file inside a large tree must not block the
# thousands of good ones around it.
# A directory the user NAMED that could not be read is a named failure, not a skip.
# `expand_directories` stamped that on each Skip; deriving it a second time here would
# mean two sets that have to agree, and nothing enforcing it. An unreadable directory
# found BENEATH a named one stays a skip: the user did not name it, so it is discovered,
# and discovered failures warn and skip.
named_failures: List[tuple[Path, Exception]] = [
(skip.path, skip.reason) for skip in skips if skip.named
]
skips = [skip for skip in skips if not skip.named]
# An unreadable named directory never became an Entry, so it has to be added to the
# count of paths considered; otherwise a lone one reports "1 of 0".
n_considered = len(entries) + len(named_failures)
to_stage: List[Path] = []
for entry in entries:
error = validate_source_path(entry.path, root, staging_root)
if error is None:
to_stage.append(entry.path)
elif entry.named:
named_failures.append((entry.path, error))
else:
logger.warning(
"%srimport: skipping '%s': %s", INDENT, entry.path, reason_text(error)
)
skips.append(Skip(entry.path, error))
if named_failures:
logger.error(
"rimport: %d of %d item(s) failed pre-flight validation; nothing was published:",
len(named_failures),
n_considered,
)
for path, error in named_failures:
logger.error("%srimport: '%s': %s", INDENT, path, reason_text(error))
return 2
# Execute the new action per file
errors = 0
for p in to_stage:
logger.info("'%s':", p)
try:
stage_data(p, root, staging_root, args.check)
except Exception as e: # pylint: disable=broad-exception-caught
# General Exception keeps CLI robust for batch runs
errors += 1
logger.error("%srimport: error processing %s: %s", INDENT, p, e)
if not errors and not args.check:
logger.info("\nNo need to run relink.py")
# Last, so it cannot be scrolled past.
report_skips(skips)
# Precedence 2 > 1 > 3 > 0; 2 already returned above. A real staging failure must never
# be masked by "completed with skips".
if errors:
return 1
if skips:
return 3
return 0
if __name__ == "__main__":
raise SystemExit(main())