-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathupdate_list.py
More file actions
3182 lines (2778 loc) · 153 KB
/
Copy pathupdate_list.py
File metadata and controls
3182 lines (2778 loc) · 153 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
# update_list.py - OmenServe-style layout, generating both lists (part 1 of 2)
import os
import io
import re
import sys
import shutil
import sqlite3
import datetime
import subprocess
import tempfile
import zipfile
import time
import json
import collections
# This script is an entry point of its own (README/INSTALL.md say "python3
# update_list.py"), and stays at the repository root for that reason - but
# everything it imports below now lives in src/ (#959).
sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "src"))
import defaults as config
import library
import platform_compat
import audio_info
# BEFORE ANYTHING PRINTS A FILENAME. This runs as its own process - the daemon
# starts it with subprocess.run() and configure.py runs it directly - so
# oserve.py's guard does nothing for it, and every line it writes is a path off
# somebody's disk.
#
# On a console whose code page cannot represent a character in one of those
# paths, print() raises UnicodeEncodeError and the scan dies where it stood.
# Reported from a live Greek-Windows install: "External update_list.py failed
# (Exit Code 1): Unknown script error", with the parent's own reader thread
# then dying on the bytes that had made it out - so the operator got a failure
# with no cause, for a library containing an accented filename.
platform_compat.install_console_encoding_guard()
# Multi-disc/box-set container names the !rar album list truncates at - see
# generate_master_list()'s own comment on the box-word block for why this
# has to match a whole PATH SEGMENT, not a substring anywhere in the path.
# An optional trailing number covers "CD1", "Disc 2", "Volume III" (digits
# only - "III" survives as part of the folder name, same as before this fix).
_BOX_WORD_RE = re.compile(
r'^(cd|disc|disk|volume|digital media|media)\s*\d*$', re.IGNORECASE)
def format_size_human(bytes_size):
for unit in ['B', 'KB', 'MB', 'GB', 'TB']:
if bytes_size < 1024.0:
return f"{bytes_size:.1f}{unit}"
bytes_size /= 1024.0
return f"{bytes_size:.1f}PB"
def format_total_size(bytes_size):
for unit in ['B', 'KiB', 'MiB', 'GiB', 'TiB']:
if bytes_size < 1024.0:
return f"{bytes_size:.1f}{unit}"
bytes_size /= 1024.0
return f"{bytes_size:.1f}PiB"
def _extension_set(setting_name):
"""One setting's extensions, normalised: lower-case and dot-leading.
Normalised HERE rather than trusted from the setting, because these are
reached from three directions - settings.conf, admin_config.py and the
dashboard's Settings page - and only one of them goes anywhere near a
validator. An operator writing "DB, .ini, tmp" means the obvious thing,
and so does "MKV,.mp4 , avi".
The leading dot is not cosmetic. `"Thumbs.db".endswith("db")` is already
true, so it is not what makes a file match; it is what stops an extension
matching the END OF A NAME. Without it, ignoring "ts" also hides every
file called `credits` or `highlights`.
An empty result is a real answer for every one of these sets, so there is
no fallback anywhere: skip nothing, no video, nothing packable.
"""
raw = getattr(config, setting_name, None)
if isinstance(raw, str):
raw = raw.split(",")
cleaned = []
for item in (raw or []):
text = str(item).strip().lower()
if not text:
continue
cleaned.append(text if text.startswith(".") else "." + text)
return tuple(dict.fromkeys(cleaned))
def ignored_extensions():
"""Extensions left out of every list."""
return _extension_set("LIST_IGNORED_EXTENSIONS")
def video_extensions():
"""Extensions routed to the film and series list rather than the music one."""
return _extension_set("LIST_VIDEO_EXTENSIONS")
def rar_extensions():
"""Extensions that make a folder packable with !rar."""
return _extension_set("RAR_EXTENSIONS")
def video_companion_extensions():
"""Extensions that follow a video into its list when they share its
folder - subtitles, .nfo, .sfv. See LIST_VIDEO_COMPANION_EXTENSIONS."""
return _extension_set("LIST_VIDEO_COMPANION_EXTENSIONS")
def pack_size_over(path, cap):
"""Is packing `path` going to exceed `cap` bytes? Returns (over, measured).
RECURSIVE, because the pack is: `rar a <dir>` takes the directory and
everything under it, so a check that only looked at the top level would
measure something other than what gets packed.
STOPS AS SOON AS THE CAP IS PASSED. The answer wanted here is a yes or no,
not a total, and the folder this is most useful on is the one that is
enormous - so walking all of it to produce a number nobody reads is the
one cost worth avoiding. A folder a hundred times over the cap is refused
after a few thousand entries instead of after all of them.
`cap` of 0 or less means no limit, and returns (False, 0) without touching
the disk at all: an operator who has not set one pays nothing for this.
An unreadable entry is skipped rather than counted or raised on. This runs
on the REQUEST path, where the alternative to an answer is a user who gets
no reply - and the pack that follows would meet the same unreadable file
and report it properly.
"""
if not cap or cap <= 0:
return False, 0
measured = 0
for root, _dirs, files in os.walk(platform_compat.long_path(path)):
for name in files:
try:
measured += os.path.getsize(
platform_compat.long_path(os.path.join(root, name)))
except OSError:
continue
if measured > cap:
return True, measured
return False, measured
def walk_with_sizes(top, onerror=None, workers=None):
"""Every file under `top`, with the size the directory entry already knew.
WHY THIS EXISTS. os.walk is built on os.scandir, which gets each entry's
size from the directory enumeration itself - and then throws it away,
because os.walk's contract is names only. The caller then asks
os.path.getsize() for a number the filesystem has just finished telling
us. One redundant syscall per file, and on a network share one redundant
ROUND TRIP per file.
Measured on 20,000 files, local SSD, warm cache, both producing the same
answer:
os.walk + getsize 0.356s (17.8 us/file)
os.scandir + cached 0.095s ( 4.7 us/file) 3.8x
That is the WALK. A whole rebuild, timed the same way over 30,000 files
and producing byte-identical lists, goes 2.51s -> 1.56s: 1.61x, because
writing and packing are the rest of the job and this does not touch them.
The walk's share is what grows on a network drive, where the second ask is
a round trip rather than a cached answer.
Local disk is the BEST case for the old shape. The library this was
written for is 799,438 files on a mapped network drive, where a rebuild
takes fifteen and a half minutes.
Yields (dirpath, [(name, size), ...]), one tuple per directory, so the
caller's loop keeps its shape.
A file whose size cannot be read yields a size of None rather than being
dropped here. That decision belongs to the caller, which already logs the
path and excludes it - see the [LIST-GEN ERROR] branch - and a helper that
silently skipped it would take that log line away.
`entry.stat()` FOLLOWS symlinks, exactly as os.path.getsize() did, so a
symlinked track still reports its target's size. On Windows that costs a
syscall only when the entry really is a symlink; an ordinary file is
answered from what the enumeration already returned.
A symlinked DIRECTORY is classified as a directory - `entry.is_dir()`,
following, exactly as os.walk does when deciding what goes in `dirs` - and
then not descended into, which is os.walk's followlinks=False default. Two
decisions, made separately, because collapsing them into
is_dir(follow_symlinks=False) answers False for a symlinked directory and
hands it back as a file.
`onerror` is called with the OSError, matching os.walk's parameter of the
same name, so an unreadable subtree is reported the way it always was
rather than ending the scan.
SEVERAL DIRECTORIES AT ONCE (#922). On a network mount every scandir()
and every entry.stat() is a round trip - Linux's d_type gives the type
but not the size - and one directory at a time, none of them overlap: a
64,136-file NFS library spent about 80 s here, with every search paused.
`workers` directories (LIST_SCAN_THREADS unless given) are listed and
stat'd at once, the way QuickList - OmenServe's list maker - walks. Each
directory is classified exactly as below whichever thread lists it; only
the order directories come back in changes, and the caller sorts before
writing anything (#443). onerror is called here, on the caller's thread,
never from a worker. One worker is the walk as it always was.
Finished directories are collected through a queue (#1124): each one's
future puts itself on it when done, and the caller takes them off one at
a time. Waiting with concurrent.futures.wait() on every outstanding
directory instead rescanned - and locked - all of them on each
completion, which made the walk quadratic in the number of directories:
41 s against 3 s on 137k files, and 16 workers slower than one.
"""
if workers is None:
workers = scan_workers()
workers = max(1, int(workers))
def list_one(current):
"""(files, subdirs, errors) for one directory, or None for files when
it could not be listed at all. Run by whichever thread gets it; says
nothing itself - the errors go back to the caller's thread."""
errors = []
subdirs = []
try:
with os.scandir(current) as scanning:
entries = list(scanning)
except OSError as err:
return None, subdirs, [err]
files = []
for entry in entries:
try:
# CLASSIFYING AND DESCENDING ARE TWO DECISIONS, and os.walk
# makes them separately. A symlink to a directory IS a
# directory - os.walk puts it in `dirs`, so it never reaches a
# caller as a file - and with followlinks=False it simply is
# not descended into.
#
# Doing both with is_dir(follow_symlinks=False) collapses them
# and gets the first one wrong: that answers False for a
# symlinked directory, which made this walk treat it as a FILE
# and stat it, and would have published a directory as a
# downloadable entry in the list.
#
# Caught by CI on Linux, where a symlink can be created without
# elevation, while the same test skipped on the Windows box
# that wrote it.
if entry.is_dir():
if not entry.is_symlink():
subdirs.append(entry.path)
continue
except OSError as err:
# A directory entry that cannot even be classified. Report it
# like an unreadable subtree - it is one - rather than
# guessing it is a file and failing again on the stat.
errors.append(err)
continue
try:
files.append((entry.name, entry.stat().st_size))
except OSError:
files.append((entry.name, None))
return files, subdirs, errors
def report(errors):
if onerror is not None:
for err in errors:
onerror(err)
if workers == 1:
pending = [top]
while pending:
current = pending.pop()
files, subdirs, errors = list_one(current)
report(errors)
pending.extend(subdirs)
if files is not None:
yield current, files
return
import queue
from concurrent.futures import ThreadPoolExecutor
pool = ThreadPoolExecutor(max_workers=workers, thread_name_prefix="list-scan")
# Each directory's future puts (directory, future) here when it is done,
# so taking the next finished one costs the same however many are still
# outstanding (#1124). SimpleQueue does its own locking.
finished = queue.SimpleQueue()
def submit(directory):
future = pool.submit(list_one, directory)
future.add_done_callback(lambda done, path=directory: finished.put((path, done)))
try:
submit(top)
outstanding = 1
while outstanding:
current, future = finished.get()
outstanding -= 1
files, subdirs, errors = future.result()
report(errors)
for subdir in subdirs:
submit(subdir)
outstanding += len(subdirs)
if files is not None:
yield current, files
finally:
# A caller that stops early - an exception mid-scan - must not leave
# workers listing a library nobody is reading any more. The futures
# this cancels still run their done-callback and land in `finished`,
# which nobody reads any more; that is harmless.
pool.shutdown(wait=True, cancel_futures=True)
def scan_workers():
"""LIST_SCAN_THREADS, held to 1..64."""
try:
wanted = int(getattr(config, "LIST_SCAN_THREADS", 16) or 1)
except (TypeError, ValueError):
wanted = 1
return max(1, min(64, wanted))
def _has_extension(name, extensions):
return bool(extensions) and str(name).lower().endswith(tuple(extensions))
def folder_totals(rows):
"""{folder: (file count, bytes)} over (folder, name, bytes) rows."""
totals = {}
for folder, _name, size in rows:
count, total = totals.get(folder, (0, 0))
totals[folder] = (count + 1, total + (size or 0))
return totals
def folder_summary_line(count, total_bytes, human):
"""The line under a folder heading that says what the folder holds (#69):
14 files, 1.20GB
ITS OWN LINE, AFTER THE CLOSING RULE, and never on the heading itself.
The heading is not decoration: list.py's reader and dcc.py's request
resolver both take the whole heading line as the folder path, and so
does every older DCCore fetching this list - a size appended there would
resolve to a folder that does not exist on every one of them. A line
that is neither a rule nor a "!" request row is skipped by all of those
readers (list.py's state machine drops it in its resting state; dcc.py
only looks at prefix lines and "!" lines; AutoQ imports "!" rows), so
this can say anything. Indented so it reads as belonging to the heading
above rather than as a heading of its own.
"""
return f" {count:,} file{'' if count == 1 else 's'}, {human(total_bytes)}"
def is_video_file(name, video=None):
"""Is this file a video, by extension?"""
return _has_extension(name, video_extensions() if video is None else video)
def is_video_companion_file(name, companions=None):
"""Is this a file that belongs beside a video - a subtitle, .nfo, .sfv?"""
return _has_extension(name, video_companion_extensions() if companions is None else companions)
def belongs_in_video_list(name, folder_has_video, video=None, companions=None):
"""Does this file go in the video list rather than the music one?
A video does, wherever it is. A companion file does only when its folder
holds a video (#411): a scene release then travels whole - the .mkv, its
.srt, its .nfo and its .sfv - while an album's .nfo stays with the album.
`folder_has_video` is the caller's, decided once per folder from the
folder's own files, so the answer for a companion cannot depend on the
order the files were met in.
"""
if is_video_file(name, video):
return True
return bool(folder_has_video) and is_video_companion_file(name, companions)
def is_packable_file(name, packable=None):
"""Does this file make its folder requestable with !rar?
Its own set, deliberately not "anything the scan indexed" - see
RAR_EXTENSIONS in defaults.py for what that cost.
"""
return _has_extension(name, rar_extensions() if packable is None else packable)
def has_backslash_component(relative_path, separator=None):
"""Does any component of `relative_path` carry a literal backslash?
The list writes folder headings with backslashes between the components
and list.list_heading_parts() reads them back by splitting on both
separators, so a backslash INSIDE a name is indistinguishable from the
separator between two - and the folder a heading resolves to is then a
different one, or none (#464).
THE ANSWER DEPENDS ON WHAT THE SEPARATOR IS, which is the whole of the
platform question here. On Linux the separator is "/" and a backslash is
an ordinary filename character, so "Rock\\Metal" is ONE folder with an
awkward name. On Windows the backslash IS the separator, so the same text
is two folders and no name can contain one - this returns False for every
path a Windows install can produce, correctly.
`separator` is for the tests, the way irc.resolve_dcc_address()'s
`lookup` is: the hazard exists on one platform, and a test that could only
run there would be a hole on the other. Production passes nothing and gets
os.sep, which is what generate_master_list() builds rel_dir with.
"""
sep = os.sep if separator is None else separator
flattened = str(relative_path).replace(sep, "/")
return any("\\" in part for part in flattened.split("/"))
def relative_folder(root, scan_root, paths=None):
"""os.path.relpath(root, scan_root), for a `root` the walk produced.
THE WALK BUILDS EVERY ROOT FROM scan_root ITSELF (#1138): each directory
is its parent's path, a separator and the entry's name, so the relative
path is simply what follows scan_root - a slice. relpath() made each one
absolute and normalised both paths first, about 10-20 us a directory on
Windows for an answer the string already held.
The slice is taken only where relpath() could not answer differently:
`root` must begin with scan_root and a separator, and what follows must
be plain names - nothing empty, no "." or "..", and on Windows no forward
slash, no colon (a drive or a stream to relpath(); no Windows name holds
one) and no name ending in a dot or a space, which Windows'
normalisation strips. Anything else asks relpath(), exactly as before.
`paths` is for the tests, like has_backslash_component()'s `separator`:
ntpath or posixpath, so both platforms' rules are checked on either.
Production passes nothing and gets os.path.
"""
paths = os.path if paths is None else paths
if root == scan_root:
return "."
sep = paths.sep
prefix = scan_root if scan_root.endswith((sep, paths.altsep or sep)) else scan_root + sep
if root.startswith(prefix):
rest = root[len(prefix):]
names = rest.split(sep)
if paths.altsep:
plain = paths.altsep not in rest and ":" not in rest and all(
name and name[-1] not in ". " for name in names)
else:
plain = all(name and name != "." and name != ".." for name in names)
if plain:
return rest
return paths.relpath(root, scan_root)
def is_listed_file(name, ignored=None):
"""Does this file go into the list? Everything does, unless it is skipped.
`ignored` is the hot-path argument: the scan resolves the setting ONCE and
passes the result down, because this is asked of every file in the library
and a 719k-file library would otherwise rebuild the tuple 719,000 times.
Callers with one file to check can leave it out.
A file with no extension is listed. So is a dotfile, and so is anything
else the operator has not named - "every file" is the rule, and the
setting is the only exception to it.
"""
if ignored is None:
ignored = ignored_extensions()
if not ignored:
return True
return not str(name).lower().endswith(ignored)
def _one_line(text):
"""Flatten anything that would break the one-entry-per-line format.
POSIX filenames may contain newlines and other control characters - only "/"
and NUL are forbidden - so a track called "evil\nname.flac" is perfectly
legal on the Linux box this daemon runs on. Written straight into the list it
splits one request entry into two lines, leaving a truncated entry and an
orphan fragment, and the file every user downloads is malformed from there
down. Windows refuses such names at creation, which is why CI caught this on
the ubuntu jobs only.
Nobody can plant one remotely - the library is the operator's own mount - so
this is robustness rather than a security boundary. Replacing with a space
keeps the entry visible and the file structurally sound; such a track is
already unrequestable, because the request parser splits on whitespace too.
Also sanitises non-UTF-8 bytes. os.walk() on POSIX decodes filenames with
the "surrogateescape" error handler by default, so a name with bytes that
are not valid UTF-8 (a CP1252 rip, a bad extraction, a FAT copy) comes back
as a lone surrogate codepoint - not itself a control character, but not
valid UTF-8 either, and this text is about to be written with a strict
UTF-8 encoder. Sanitised here rather than left to fail at the write: one
bad name in a library of thousands now costs a mangled-but-valid name in
the list, not the entire rebuild.
Almost no name needs either change, so one regex search decides first
(#1125): the per-character pass below cost about 15 us per row, and the
largest share of a rebuild's writing time. The pattern is exactly the
characters the slow path changes - the controls and DEL it flattens, and
the lone surrogates that the UTF-8 'replace' round trip turns into "?",
which is the only thing that round trip changes.
"""
text = str(text)
if _NEEDS_ONE_LINE_CLEANING.search(text) is None:
return text
text = text.encode("utf-8", "replace").decode("utf-8")
return "".join(" " if ch < " " or ch == "\x7f" else ch for ch in text)
# What _one_line() would change: a control character, DEL, or a lone
# surrogate (#1125). A name with none of them is returned as it is.
_NEEDS_ONE_LINE_CLEANING = re.compile("[\x00-\x1f\x7f\ud800-\udfff]")
def _heading_text(folder):
"""The heading line a folder gets in the list, exactly as written.
One function for the music list and for the rewrite of its length and
quality (#1182), which finds each row's cache entry by the heading above
it: two copies of this expression could drift, and a row would then miss
its suffix in one and not the other."""
import list as list_mod
raw = f"{list_mod.LIST_FOLDER_PREFIX}{folder}\\" if folder else list_mod.LIST_FOLDER_PREFIX
return _one_line(raw.replace("/", "\\"))
def _discard_temp_lists(*paths):
"""Remove half-written temporary lists so they cannot be mistaken for real ones."""
for path in paths:
try:
if path and os.path.exists(path):
os.remove(path)
except OSError as err:
print(f"[LIST-CLEAN ERROR] Could not remove {path}: {err}")
# How patient the swap is with a reader holding a list open (#923). Searches
# now run during the scan; the swap refuses new ones, but one that started a
# moment before may still be reading the list - a second or two on a very
# large one - and on Windows the rename waits for it. Ten attempts back off to
# about ten seconds in all; POSIX never retries at all.
PUBLISH_REPLACE_ATTEMPTS = 10
def _publish_artifacts(swaps):
"""Move every (temporary, destination) pair into place, or none of them.
THE PUBLISH USED TO BE FIVE INDEPENDENT REPLACEMENTS, so a failure partway
through left the bot advertising one scan and handing out another. The
ordinary way to reach it, on Windows: os.replace onto a file another
handle has open raises PermissionError, and dcc.py holds the published
artifact open for the whole duration of a DCC send. PAUSE_ON_UPDATE only
refuses NEW requests, so a transfer already in flight keeps that handle -
and somebody downloading the list when the scheduled rebuild lands is not
an edge case on a bot with several slots.
What that produced: the master index replaced, so @find, the advert count
and commands.count_from_master_list() all reported the new scan - while
the archive users actually received was the previous one, the size side
files still carried the previous numbers, the base-name marker was not
updated, and the prune never ran. Nothing recovered it; the artifact
stayed stale until some later rebuild happened to run with no transfer in
progress. And the failure branch then printed "The previous list was left
untouched and is still in use", which by then was false.
Each destination is moved ASIDE before its replacement lands, so a failure
can put back exactly what was there. That also makes the locked case fail
at the safest possible moment: renaming a file another process holds open
fails on Windows too, so the lock is discovered while moving the old file
out of the way - before anything observable has changed.
Raises whatever the underlying replace raised, after rolling back. The
caller's message about the previous list still being in use is then true
again, which is the point.
"""
done = []
try:
for temporary, destination in swaps:
backup = None
if os.path.exists(platform_compat.long_path(destination)):
backup = destination + ".previous"
platform_compat.replace_with_retry(destination, backup,
attempts=PUBLISH_REPLACE_ATTEMPTS)
platform_compat.replace_with_retry(temporary, destination,
attempts=PUBLISH_REPLACE_ATTEMPTS)
done.append((destination, backup))
except Exception:
# Reverse order because that is the convention for undoing a
# sequence, not because it is required here: each swap touches only
# its own destination and that destination's .previous, so no two of
# them can collide. A mutation run flipped the order and nothing
# failed, which is the honest reading - the comment that used to sit
# here claimed a necessity there is not one of.
for destination, backup in reversed(done):
try:
if backup:
platform_compat.replace_with_retry(backup, destination)
else:
# There was nothing here before; leaving the new file
# would publish half a rebuild.
os.remove(platform_compat.long_path(destination))
except OSError as undo_err:
print(f"[LIST-GEN ERROR] Could not roll {destination} back: "
f"{undo_err}")
raise
for _destination, backup in done:
if not backup:
continue
try:
os.remove(platform_compat.long_path(backup))
except OSError:
# A leftover .previous is clutter, not a failure - the publish
# itself succeeded and that is what the caller is waiting on.
pass
return True
def _prune_superseded_lists(keep, directory=None):
"""Delete older generated lists once the new ones are safely in place.
Runs AFTER the swap, never before: the previous index has to stay usable for the whole
scan, which can take minutes on a large NFS mount.
"""
# The list's OWN directory. Every list prunes only what it wrote: with
# more than one, a prune reaching across them would delete another list's
# current index the moment their base names matched, which they always do.
directory = directory or config.LOCAL_LIST_DIR
removed = 0
try:
entries = os.listdir(directory)
except OSError as err:
print(f"[LIST-CLEAN ERROR] Could not read {directory}: {err}")
return
for item in entries:
if item in keep:
continue
# The HYPHEN is the point. Every generated list is
# f"{LIST_BASE_NAME}-{today}.txt" or f"{LIST_BASE_NAME}-RAR-{today}.txt",
# so the separator is always there - and matching on the bare prefix
# also matched the side files, which live in this same directory.
#
# With LIST_BASE_NAME derived from a nickname like "dccore" or "dcc",
# the size side file starts with that prefix too, so every rebuild
# wrote the side files and then deleted them again. The library's size
# then disappeared from every public surface permanently: the advert
# published "Files (0B)" and the CTCP SLOTS payload published 0 raw
# bytes, on every interval, for ever - and the log line for it read
# "[LIST-CLEAN] Removed 2 superseded list(s)", which sounds like
# housekeeping working. Found by audit.
if not item.startswith(config.LIST_BASE_NAME + "-"):
continue
# Belt and braces: never remove a file this run just wrote, whatever
# the name matching decides.
if item in (os.path.basename(str(getattr(config, "LIST_SIZE_FILE", ""))),
os.path.basename(str(getattr(config, "LIST_RAWBYTES_FILE", "")))):
continue
# ".rar" is here because LIST_FORMAT can publish one. Only names that
# also start with LIST_BASE_NAME are considered, and this is the lists
# directory, so no album archive a user is waiting on is in reach.
if not item.endswith((".txt", ".zip", ".rar")):
continue
try:
os.remove(os.path.join(directory, item))
removed += 1
except OSError as err:
print(f"[LIST-CLEAN ERROR] Could not remove {item}: {err}")
if removed:
print(f"[LIST-CLEAN] Removed {removed} superseded list(s).")
# The literal shipped default defaults.py derives LIST_BASE_NAME away FROM -
# see its own comment on why the derivation compares against this same
# literal rather than a shared constant (a snapshot taken before overrides
# apply, the way SHIPPED_DEFAULTS does for settings_file.REQUIRED, would
# need LIST_BASE_NAME added to REQUIRED just to get one, which it deliberately
# is not).
_SHIPPED_LIST_BASE_NAME = "DCCore"
# The name the artifacts in LOCAL_LIST_DIR were last published under.
#
# #213: the migration below could only ever carry files across from the shipped
# default, so it worked exactly once. Rename the bot a second time - or from any
# value that was never "DCCore" - and the artifacts on disk keep the old name
# while the bot looks for the new one. A restart does not recover it: the list
# is right there and invisible, and the advert says 0 files.
#
# The previous name has to be remembered somewhere. A marker file in the lists
# directory rather than settings.conf/admin_config.py, because it travels with
# the thing it describes: a value in the config can be hand-edited, replaced
# wholesale on an upgrade, or restored from a backup taken before the rename,
# and each of those silently orphans the lists again - which is the exact
# failure being closed. A file in the directory cannot drift from the directory
# it names, because it is in it.
#
# Absent means an install from before this existed, which is precisely when
# falling back to _SHIPPED_LIST_BASE_NAME is the right guess.
#
# The leading dot and the lack of a .txt/.zip/.rar suffix keep it out of every
# artifact scan in list.py and out of the glob in find_latest_list().
_LIST_BASE_MARKER = ".dccore-list-base"
def list_base_marker_path(directory=None):
directory = directory or getattr(config, "LOCAL_LIST_DIR", "./lists")
return os.path.join(directory, _LIST_BASE_MARKER)
def read_list_base_marker(directory=None):
"""The LIST_BASE_NAME the artifacts on disk were published under, or None.
None on any read problem, deliberately: an unreadable marker must fall back
to the shipped-default guess, never raise into startup.
"""
try:
with io.open(list_base_marker_path(directory), encoding="utf-8") as handle:
return handle.read().strip() or None
except (OSError, UnicodeDecodeError):
return None
def write_list_base_marker(name=None, directory=None, log=print):
"""Record the name the artifacts are now published under. Returns True on
success.
Written through db._atomic_write, like the advert side files: a marker
truncated by a crash mid-write would read as absent, sending the next
migration back to the shipped-default guess and stranding the artifacts
this call exists to keep findable.
"""
import db
name = config.LIST_BASE_NAME if name is None else name
try:
db._atomic_write(list_base_marker_path(directory), str(name) + "\n")
return True
except Exception as err:
log(f"[MIGRATE] Could not record the list base name: {err}. "
f"A future rename may not find these files.")
return False
def migrate_list_base_name(log=print):
"""Carry existing list files across when LIST_BASE_NAME changed out from
under them - #184's review: defaults.py's LIST_BASE_NAME
derivation (an untouched value takes NICKNAME's own value once NICKNAME
is set) means every existing install's list files, generated before that
derivation existed, are sitting on disk as "DCCore-<date>.*" while
LIST_BASE_NAME now resolves to the operator's nickname instead.
Without this, find_latest_list() globs for the NEW base name, finds
nothing, and the daemon boots, joins its channels and advertises with no
list at all - not because there is no list, but because the one on disk
is filed under a name nothing is looking for any more. It stays that way
until the next successful !update, which on a weekly rebuild schedule is
up to a week of a bot that looks healthy and answers every request with
"not found". The side-file migration's own docstring described
this exact failure shape for the side-file rename; this is the same
problem, for a prefix rather than a single filename.
Deliberately narrow, matching that function's safety properties:
* only when LIST_BASE_NAME no longer equals what defaults.py ships -
an install that never had any "DCCore-*" files (fresh, or one that
chose its own LIST_BASE_NAME from the very first run) has nothing to
move, and this is a no-op for it.
* only a file whose new name does not already exist is moved - a
rebuild that has already happened under the new name wins over
anything left behind from the old one.
* os.replace, so an interrupted run leaves one intact file rather than
two halves; a failure is logged and swallowed per file, because a
daemon that will not start over a rename is a worse outcome than the
rename not happening for one file.
Returns the list of (old, new) basenames actually moved, for the tests
and for the startup log.
"""
# EVERY list's directory, not only the primary's. A list's files live in
# its own directory - list.list_dir(name), with the primary keeping
# LOCAL_LIST_DIR itself - and the marker that records what they are called
# is already per-directory. This function simply never looked anywhere but
# the primary, so renaming the bot orphaned every other list's artifacts:
# they kept the old base name, nothing on the next startup knew to look
# for it, and each of those channels advertised a library it no longer had
# a list for.
#
# A failure in one directory must not stop the others, for the same reason
# a failure on one file does not stop the rest of that directory: a daemon
# that will not start over a rename is worse than the rename not
# happening.
import list as list_mod
import library
# No de-duplication of directories. It was written, and then deleted for
# being unreachable: list_dir() answers LOCAL_LIST_DIR for any list marked
# primary, so two primaries would share a directory - but load_lists()
# normalises a hand-edited file down to exactly one primary before this
# ever sees it. A guard that cannot be reached is a guard no test can
# falsify, and this file has deleted two of those already.
moved = []
for served in library.lists():
try:
directory = list_mod.list_dir(served.name)
except Exception as err: # a malformed lists.json is not fatal here
log(f"[MIGRATE] Could not resolve the directory for list "
f"{served.name!r}: {err}")
continue
moved.extend(_migrate_one_list_directory(directory, log=log))
return moved
def _migrate_one_list_directory(directory, log=print):
"""migrate_list_base_name() for ONE list's directory. Returns the
(old, new) basenames actually moved."""
# What the files on disk are actually called, not what they were called
# when the bot shipped. Absent means an install from before the marker
# existed, where the shipped default is the right guess (#213).
previous = read_list_base_marker(directory) or _SHIPPED_LIST_BASE_NAME
if previous == config.LIST_BASE_NAME:
# Nothing to move. Still record it: an install that has never been
# renamed has no marker, and writing one now means its FIRST rename is
# migrated from the right name rather than from the shipped guess.
write_list_base_marker(config.LIST_BASE_NAME, directory, log=log)
return []
try:
entries = os.listdir(directory)
except OSError as err:
log(f"[MIGRATE] Could not read {directory}: {err}")
return []
# The "-" is required, not just the bare prefix: every real artifact is
# named "DCCore-<date>.ext" or "DCCore-RAR-<date>.ext", always with a
# hyphen immediately after the base name. A bare startswith("DCCore")
# would also match a file already renamed to the NEW LIST_BASE_NAME when
# that name itself happens to start with "DCCore" - e.g. "DCCoreTest" -
# corrupting an already-correct file instead of leaving it alone.
old_prefix = previous + "-"
moved = []
for item in entries:
if not item.startswith(old_prefix):
continue
if not item.endswith((".txt", ".zip", ".rar")):
continue
new_name = config.LIST_BASE_NAME + item[len(previous):]
old_path = os.path.join(directory, item)
new_path = os.path.join(directory, new_name)
if os.path.exists(new_path):
continue
try:
platform_compat.replace_with_retry(old_path, new_path)
moved.append((item, new_name))
except OSError as err:
log(f"[MIGRATE] Could not rename {item} to {new_name}: {err}. "
f"It will not be found until the next successful !update.")
for old_name, new_name in moved:
log(f"[MIGRATE] Renamed {old_name} to {new_name}.")
# After the move, not before: if the renames failed the marker must still
# say what the files on disk are really called, or the next startup would
# look for them under a name nothing has.
if moved or not read_list_base_marker(directory):
write_list_base_marker(config.LIST_BASE_NAME, directory, log=log)
return moved
def _artifact_paths(fmt, date_str, directory=None, staging=".new"):
"""Where the download artifact for `fmt` is published, and staged. The
audio reading's rewrite stages under its own suffix (#1182).
"""
import list as list_mod
final = os.path.join(directory or config.LOCAL_LIST_DIR,
list_mod.list_artifact_name(fmt, date_str))
return final, final + staging
def _write_text_artifact(tmp_path, members):
"""Both lists as one text file.
A plain .txt can only be one file where the .zip is two, so the album
section is appended to the file list rather than dropped. The format is a
choice about packaging; it is not a request to hand out less. This is a
copy and not the master index itself precisely because the two must not be
the same file - see list.FULL_LIST_MARKER.
THE OPERATOR'S BANNER APPEARS ONCE, NOT ONCE PER SECTION
Each source list carries its own banner, which is right when they are
downloaded separately - as .zip and .rar hand them out, and as the !rar
list is served on its own. Concatenated, that put the operator's ASCII art
in the middle of the file as well as at the top, which reads as a bug
rather than as a design.
The identity line deliberately still repeats. It is a section header - the
album half of this file should say what is serving it too, and one line is
not noise. The banner can be any height, which is the difference.
Removed by matching the exact text read_operator_header() returned rather
than by recognising a banner in the output, because only the first is
knowable: the banner is free-form and could otherwise be anything,
including something that looks like a folder heading.
"""
banner = read_operator_header()
# Exactly as generate_master_list() wrote it: a leading blank line, the
# banner, then the newline ending its last line.
banner_block = "\n" + banner + "\n" if banner else ""
with io.open(tmp_path, "w", encoding="utf-8", newline="\n") as out:
for index, (source, _name) in enumerate(members):
if index:
out.write("\n\n")
_copy_removing_first(source, out,
banner_block if index else "")
# A megabyte at a time. Large enough that the read count is irrelevant next to
# the work of writing, small enough to be noise beside what the scan is
# already holding.
_COPY_CHUNK_CHARS = 1 << 20
def _copy_removing_first(source, out, strip=""):
"""Copy `source` into `out`, removing the FIRST occurrence of `strip`.
In chunks (#463). This was `handle.read()` followed by
`text.replace(strip, "", 1)`, and the file it reads is a list - several
hundred megabytes on a library big enough for the operator to have chosen
the txt format in the first place, held whole in memory on top of
everything the rebuild is already holding.
THE OVERLAP IS WHAT KEEPS IT FAITHFUL. A chunk boundary can fall inside
the text being removed, so the last len(strip) - 1 characters of each
chunk are held back and carried into the next rather than written out.
Without that, a banner straddling a boundary would be written through -
and the result would be a list with the operator's banner repeated in the
middle of it, which is the exact thing this removal exists to prevent.
"""
pending = ""
with io.open(source, encoding="utf-8") as handle:
while True:
chunk = handle.read(_COPY_CHUNK_CHARS)
if not chunk:
break
if not strip:
out.write(chunk)
continue
pending += chunk
at = pending.find(strip)
if at >= 0:
out.write(pending[:at] + pending[at + len(strip):])
pending = ""
strip = ""
continue
carry = len(strip) - 1
if carry:
out.write(pending[:-carry])
pending = pending[-carry:]
else:
out.write(pending)
pending = ""
if pending:
out.write(pending)
def _write_zip_artifact(tmp_path, members):
"""Store the temp files under their FINAL names, so the archive users
download is identical to what it always was."""
# One more heartbeat before the longest silent step in the run.
# Deflating a list this size is minutes on a machine that has just
# walked 80 TB, and the daemon's stall watch has nothing else to go
# on until it finishes - see commands.run_watching_for_a_stall().
write_progress("packing", force=True)
with zipfile.ZipFile(tmp_path, "w", zipfile.ZIP_DEFLATED) as zipf:
for source, name in members:
zipf.write(source, arcname=name)
_STAGING_PREFIX = ".listpack-"
# How often a running scan may write its progress file. A 719k-file library
# would otherwise spend a meaningful part of the run serialising JSON nobody
# read - the dashboard polls every couple of seconds, so anything finer is
# work done for no reader.
PROGRESS_WRITE_SECONDS = 0.5
_progress_last_write = [0.0]
# WHEN THIS RUN BEGAN, stamped once and repeated in every write.
#
# The dashboard cannot work it out for itself: update_list.py is a SUBPROCESS,