[scrobbles] Add dedup_scrobbles command to remove duplicate scrobbles (69f22940)
All checks were successful
ci / test (push) Successful in 2m6s
ci / build-and-deploy (push) Has been skipped

This commit is contained in:
2026-08-10 00:31:54 -04:00
parent cbf027203c
commit bce6639b04
4 changed files with 226 additions and 4 deletions

View File

@ -627,7 +627,18 @@ The Edit log form should have from top to bottom:
- Expansion ids (which should a multi-select widget of expansions for this game) - Expansion ids (which should a multi-select widget of expansions for this game)
- Location (which should be a drop down of BoardGameLocations for this user) - Location (which should be a drop down of BoardGameLocations for this user)
** TODO [#A] Dedup track scrobbles from lastfm import :importers:lastfm:tracks: ** DONE [#A] Dedup track scrobbles from lastfm import :importers:lastfm:tracks:
:PROPERTIES:
:ID: 69f22940-8cee-4a82-b8ea-3d3d770f5e1b
:END:
*** Description
The historical LastFM import duplicated a number of scrobbles. The importer
itself was not fixed, but a `dedup_scrobbles` management command now removes
duplicate scrobbles where the start timestamp, end timestamp, and media
scrobbled are identical. Run it dry (default) to report duplicates, or with
`--commit` to delete them. A `--media-type` filter limits the scan.
** DONE [#A] Write a script to check for raw mopidy vs found track discrepancies :tracks:metadata: ** DONE [#A] Write a script to check for raw mopidy vs found track discrepancies :tracks:metadata:
:PROPERTIES: :PROPERTIES:
:ID: 0b98a1b9-2848-433e-8bec-0f5737cfc74b :ID: 0b98a1b9-2848-433e-8bec-0f5737cfc74b

View File

@ -1,10 +1,145 @@
from datetime import datetime from datetime import datetime
import pytz import pytz
from django.contrib.auth import get_user_model from django.contrib.auth import get_user_model
from django.utils import timezone
from music.models import Artist, Track
from scrobbles.models import Scrobble
from vrobbler.apps.scrobbles.utils import timestamp_user_tz_to_utc from vrobbler.apps.scrobbles.utils import (
deduplicate_scrobbles,
timestamp_user_tz_to_utc,
)
User = get_user_model()
def test_timestamp_user_tz_to_utc(): def test_timestamp_user_tz_to_utc():
timestamp = timestamp_user_tz_to_utc(1685561082, pytz.timezone("US/Eastern")) timestamp = timestamp_user_tz_to_utc(1685561082, pytz.timezone("US/Eastern"))
assert timestamp == datetime(2023, 5, 31, 23, 24, 42, tzinfo=pytz.utc) assert timestamp == datetime(2023, 5, 31, 23, 24, 42, tzinfo=pytz.utc)
def _make_track():
artist = Artist.objects.create(name="Carly Rae Jepsen")
return Track.objects.create(
title="Emotion",
artist_fk=artist,
base_run_time_seconds=60,
)
def test_deduplicate_scrobbles_removes_exact_duplicates(db):
user = User.objects.create(email="dup@example.com")
track = _make_track()
timestamp = timezone.now().replace(microsecond=0)
stop_timestamp = timestamp + timezone.timedelta(seconds=60)
for _ in range(3):
Scrobble.objects.create(
user=user,
track=track,
media_type="Track",
timestamp=timestamp,
stop_timestamp=stop_timestamp,
)
count = deduplicate_scrobbles(commit=True)
assert count == 2
assert Scrobble.objects.filter(user=user, track=track).count() == 1
def test_deduplicate_scrobbles_keeps_different_media(db):
user = User.objects.create(email="media@example.com")
track = _make_track()
other_track = Track.objects.create(
title="Other",
artist_fk=Artist.objects.create(name="Someone Else"),
base_run_time_seconds=60,
)
timestamp = timezone.now().replace(microsecond=0)
stop_timestamp = timestamp + timezone.timedelta(seconds=60)
Scrobble.objects.create(
user=user,
track=track,
media_type="Track",
timestamp=timestamp,
stop_timestamp=stop_timestamp,
)
Scrobble.objects.create(
user=user,
track=other_track,
media_type="Track",
timestamp=timestamp,
stop_timestamp=stop_timestamp,
)
count = deduplicate_scrobbles(commit=True)
assert count == 0
assert Scrobble.objects.count() == 2
def test_deduplicate_scrobbles_keeps_different_timestamps(db):
user = User.objects.create(email="time@example.com")
track = _make_track()
timestamp = timezone.now().replace(microsecond=0)
stop_timestamp = timestamp + timezone.timedelta(seconds=60)
Scrobble.objects.create(
user=user,
track=track,
media_type="Track",
timestamp=timestamp,
stop_timestamp=stop_timestamp,
)
Scrobble.objects.create(
user=user,
track=track,
media_type="Track",
timestamp=timestamp + timezone.timedelta(hours=1),
stop_timestamp=stop_timestamp + timezone.timedelta(hours=1),
)
count = deduplicate_scrobbles(commit=True)
assert count == 0
assert Scrobble.objects.count() == 2
def test_deduplicate_scrobbles_dry_run_does_not_delete(db):
user = User.objects.create(email="dry@example.com")
track = _make_track()
timestamp = timezone.now().replace(microsecond=0)
stop_timestamp = timestamp + timezone.timedelta(seconds=60)
for _ in range(2):
Scrobble.objects.create(
user=user,
track=track,
media_type="Track",
timestamp=timestamp,
stop_timestamp=stop_timestamp,
)
count = deduplicate_scrobbles(commit=False)
assert count == 1
assert Scrobble.objects.filter(user=user, track=track).count() == 2
def test_deduplicate_scrobbles_filters_by_media_type(db):
user = User.objects.create(email="filter@example.com")
track = _make_track()
timestamp = timezone.now().replace(microsecond=0)
stop_timestamp = timestamp + timezone.timedelta(seconds=60)
for _ in range(2):
Scrobble.objects.create(
user=user,
track=track,
media_type="Track",
timestamp=timestamp,
stop_timestamp=stop_timestamp,
)
count = deduplicate_scrobbles(commit=True, media_type="Video")
assert count == 0
assert Scrobble.objects.filter(user=user, track=track).count() == 2

View File

@ -0,0 +1,32 @@
from django.core.management.base import BaseCommand
from vrobbler.apps.scrobbles.utils import deduplicate_scrobbles
class Command(BaseCommand):
def add_arguments(self, parser):
parser.add_argument(
"--commit",
action="store_true",
help="Commit the deletes",
)
parser.add_argument(
"--media-type",
default=None,
help="Only dedupe scrobbles of this media type (e.g. Track)",
)
def handle(self, *args, **options):
commit = options["commit"]
media_type = options["media_type"]
if not commit:
self.stdout.write("No changes will be saved, use --commit to save")
count = deduplicate_scrobbles(commit=commit, media_type=media_type)
if commit:
self.stdout.write(
self.style.SUCCESS(f"Deleted {count} duplicate scrobbles")
)
else:
self.stdout.write(f"Found {count} duplicate scrobbles")

View File

@ -90,7 +90,8 @@ def get_scrobbles_for_media(media_obj, user: User) -> models.QuerySet:
return Scrobble.objects.filter(media_query, user=user) return Scrobble.objects.filter(media_query, user=user)
def get_recently_played_board_games(user: User) -> dict: ... def get_recently_played_board_games(user: User) -> dict:
...
def get_long_play_media_model(app_label: str, model_name: str): def get_long_play_media_model(app_label: str, model_name: str):
@ -240,6 +241,46 @@ def delete_zombie_scrobbles(dry_run=True):
return zombies_found return zombies_found
def deduplicate_scrobbles(commit=True, media_type=None):
"""Delete scrobbles duplicated by the Last.fm historical import.
A duplicate is any scrobble whose (user, timestamp, stop_timestamp, media)
combination matches another scrobble. The oldest scrobble in each duplicate
group is kept and the rest are deleted. Passing ``media_type`` limits the
scan to a single media type. Returns the number of duplicate scrobbles
found.
"""
from scrobbles.models import TYPE_FK_PREFETCHES, Scrobble
media_types = [media_type] if media_type else list(TYPE_FK_PREFETCHES)
group_cols = ("user_id", "timestamp", "stop_timestamp")
duplicate_ids = []
for mtype in media_types:
fk_field = TYPE_FK_PREFETCHES[mtype][0]
scrobbles = Scrobble.objects.filter(
media_type=mtype, **{f"{fk_field}__isnull": False}
)
groups = (
scrobbles.values(*group_cols, f"{fk_field}_id")
.annotate(count=models.Count("id"))
.filter(count__gt=1)
)
for group in groups:
group_filter = {col: group[col] for col in (*group_cols, f"{fk_field}_id")}
keep_id = scrobbles.filter(**group_filter).order_by("id").first().id
duplicate_ids.extend(
scrobbles.filter(**group_filter)
.exclude(id=keep_id)
.values_list("id", flat=True)
)
if duplicate_ids and commit:
Scrobble.objects.filter(id__in=duplicate_ids).delete()
return len(duplicate_ids)
def import_from_webdav_for_all_users(restart=False): def import_from_webdav_for_all_users(restart=False):
"""Grab a list of all users with WebDAV enabled and kickoff imports for them""" """Grab a list of all users with WebDAV enabled and kickoff imports for them"""
from books.koreader import fetch_file_from_webdav from books.koreader import fetch_file_from_webdav
@ -356,7 +397,10 @@ def apply_media_unit_conversions(form, media_type: str, user) -> str:
drink_types = ("Beer", "Wine", "Coffee") drink_types = ("Beer", "Wine", "Coffee")
if media_type in drink_types and "size_ml" in form.fields: if media_type in drink_types and "size_ml" in form.fields:
use_oz = getattr(getattr(user, "profile", None), "volume_unit", "metric") == "imperial" use_oz = (
getattr(getattr(user, "profile", None), "volume_unit", "metric")
== "imperial"
)
if use_oz: if use_oz:
form.fields["size_ml"].label = "Size (oz)" form.fields["size_ml"].label = "Size (oz)"
original_clean = form.fields["size_ml"].clean original_clean = form.fields["size_ml"].clean