feat(trip-cluster): cluster orchestration (seed derivation + coverage) + CLI
This commit is contained in:
@@ -0,0 +1,61 @@
|
||||
from photoflow.core import Store
|
||||
from photoflow.core.models import Asset
|
||||
from photoflow.immich import pipeline
|
||||
from app.cluster_run import run_cluster
|
||||
|
||||
|
||||
def _store(tmp_path):
|
||||
return Store(str(tmp_path / "t.db")).connect()
|
||||
|
||||
|
||||
def test_run_cluster_seeds_bounded_tag_not_people_tag(tmp_path):
|
||||
s = _store(tmp_path)
|
||||
# "Italy 2019" spans 2 days -> seeds; "Mum" spans years -> not a seed.
|
||||
rows = [("a", "2019-06-01T10:00:00", ["Italy 2019", "Mum"]),
|
||||
("b", "2019-06-02T10:00:00", ["Italy 2019"]),
|
||||
("c", "2010-01-01T10:00:00", ["Mum"]),
|
||||
("d", "2022-01-01T10:00:00", ["Mum"])]
|
||||
for i, t, tags in rows:
|
||||
s.upsert_asset(Asset(immich_id=i, taken_at=t))
|
||||
s.set_asset_tags(i, tags)
|
||||
res = run_cluster(s)
|
||||
seeded = [c for c in s.all_clusters() if c.suggested_name == "Italy 2019"]
|
||||
assert len(seeded) == 1
|
||||
assert sorted(x.immich_id for x, _ in s.cluster_members(seeded[0].id)) == ["a", "b"]
|
||||
assert res["clusters"] >= 1
|
||||
s.close()
|
||||
|
||||
|
||||
def test_run_cluster_excludes_processed_and_replaces(tmp_path):
|
||||
s = _store(tmp_path)
|
||||
s.upsert_asset(Asset(immich_id="a", taken_at="2019-06-01T10:00:00"))
|
||||
s.upsert_asset(Asset(immich_id="p", taken_at="2019-06-01T11:00:00"))
|
||||
s.set_asset_tags("a", [])
|
||||
s.set_asset_tags("p", [pipeline.PROCESSED])
|
||||
s.mark_processed("p")
|
||||
run_cluster(s)
|
||||
run_cluster(s) # idempotent replace — not doubled
|
||||
all_member_ids = [x.immich_id for c in s.all_clusters()
|
||||
for x, _ in s.cluster_members(c.id)]
|
||||
assert "p" not in all_member_ids
|
||||
assert all_member_ids.count("a") == 1
|
||||
s.close()
|
||||
|
||||
|
||||
def test_run_cluster_short_span_non_dominant_tag_does_not_seed(tmp_path):
|
||||
s = _store(tmp_path)
|
||||
# "Lunch" is on 2 photos within a ~30-min window, but that same window is
|
||||
# full of untagged photos -> the tag does NOT dominate its window
|
||||
# (2/5 = 0.4 < 0.5), so it must not seed even though its span is tiny.
|
||||
rows = [("x", "2020-03-01T12:00:00", ["Lunch"]),
|
||||
("y", "2020-03-01T12:30:00", ["Lunch"]),
|
||||
("u1", "2020-03-01T12:05:00", []),
|
||||
("u2", "2020-03-01T12:15:00", []),
|
||||
("u3", "2020-03-01T12:25:00", [])]
|
||||
for i, t, tags in rows:
|
||||
s.upsert_asset(Asset(immich_id=i, taken_at=t))
|
||||
s.set_asset_tags(i, tags)
|
||||
run_cluster(s)
|
||||
seeded = [c for c in s.all_clusters() if c.suggested_name == "Lunch"]
|
||||
assert seeded == []
|
||||
s.close()
|
||||
Reference in New Issue
Block a user