RehamAAhmed commited on
Commit
36ff019
·
verified ·
1 Parent(s): fa0a0b6

Upload 4 files

Browse files
Files changed (4) hide show
  1. furniture_matcher.py +645 -0
  2. mllm_inspector.py +139 -0
  3. report_generator.py +507 -0
  4. sam_masker.py +106 -0
furniture_matcher.py ADDED
@@ -0,0 +1,645 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os, sys, shutil
2
+ # Set HF_HOME dynamically based on available disk space
3
+ d_free = 0
4
+ try:
5
+ d_free = shutil.disk_usage('D:/').free
6
+ except Exception:
7
+ pass
8
+
9
+ if d_free > 500 * 1024 * 1024: # At least 500MB free
10
+ os.environ['HF_HOME'] = 'D:/hf_cache'
11
+ else:
12
+ os.environ['HF_HOME'] = 'C:/hf_cache'
13
+
14
+ os.environ['HF_HUB_DISABLE_SYMLINKS_WARNING'] = '1'
15
+ if sys.stdout.encoding != 'utf-8':
16
+ sys.stdout.reconfigure(encoding='utf-8', errors='replace')
17
+ if sys.stderr.encoding != 'utf-8':
18
+ sys.stderr.reconfigure(encoding='utf-8', errors='replace')
19
+ """
20
+ Furniture Inventory Matcher — High-Precision Edition
21
+ ======================================================
22
+ Problem: Given BEFORE/AFTER folders, report each item as:
23
+ ✅ FOUND — same item detected in AFTER
24
+ ❌ MISSING — not found (could be moved/stolen/removed)
25
+
26
+ Accuracy requirement: BOTH false-positives AND false-negatives must be minimal.
27
+
28
+ Architecture (3-stage cascade):
29
+ Stage 1 — DINOv2 CLS global embedding (fast, semantic)
30
+ Stage 2 — DINOv2 PATCH-level matching (robust to crops/scale/partial views)
31
+ Stage 3 — SIFT + RANSAC geometry (confirms spatial consistency)
32
+
33
+ Why this combination?
34
+ • DINOv2 CLS → understands "what" the object is (semantic identity)
35
+ • Patch match → finds correspondences even if only part of object is visible
36
+ • SIFT/RANSAC → geometric proof that pixel-level structure matches
37
+
38
+ Invariant to: lighting, scale, rotation, partial occlusion, crop differences.
39
+
40
+ Usage:
41
+ python furniture_matcher.py --before ./before --after ./after
42
+ python furniture_matcher.py --before ./before --after ./after --output ./report
43
+ python furniture_matcher.py --before ./before --after ./after --strict
44
+ """
45
+
46
+ import os, sys, time, argparse, base64
47
+ import numpy as np
48
+ from pathlib import Path
49
+ from PIL import Image
50
+
51
+ # ── Torch / DINOv2 ──────────────────────────────────────────────────────────
52
+ try:
53
+ import torch
54
+ from transformers import AutoImageProcessor, AutoModel
55
+ except ImportError:
56
+ print("[ERROR] pip install torch transformers Pillow")
57
+ sys.exit(1)
58
+
59
+ # ── OpenCV ───────────────────────────────────────────────────────────────────
60
+ try:
61
+ import cv2
62
+ CV2_OK = True
63
+ except ImportError:
64
+ CV2_OK = False
65
+ print("[WARNING] pip install opencv-python — geometric verification disabled")
66
+
67
+ # ── Rich (prettier output) ───────────────────────────────────────────────────
68
+ try:
69
+ from rich.console import Console
70
+ from rich.table import Table
71
+ from rich.panel import Panel
72
+ from rich.progress import Progress, SpinnerColumn, BarColumn, TextColumn, TaskProgressColumn
73
+ RICH = True
74
+ console = Console()
75
+ except ImportError:
76
+ RICH = False
77
+ console = None
78
+
79
+ # ═══════════════════════════════════════════════════════════════════════════
80
+ # CONFIGURATION
81
+ # ═══════════════════════════════════════════════════════════════════════════
82
+ MODEL_ID = "facebook/dinov2-base" # 86M params, ~330MB — excellent for furniture
83
+ IMAGE_EXTS = {".jpg", ".jpeg", ".png", ".bmp", ".webp", ".tiff", ".tif"}
84
+ DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
85
+
86
+ # ── Score thresholds ─────────────────────────────────────────────────────────
87
+ # Stage 1 (CLS global):
88
+ CLS_CONFIRM_FOUND = 0.88 # score ≥ this → FOUND immediately (very safe)
89
+ CLS_CONFIRM_MISSING = 0.60 # score < this → MISSING immediately (very safe)
90
+ # Between these → go to Stage 2
91
+
92
+ # Stage 2 (patch-level):
93
+ PATCH_CONFIRM_FOUND = 0.72
94
+ PATCH_CONFIRM_MISSING = 0.55
95
+ # Between → go to Stage 3
96
+
97
+ # Stage 3 (SIFT+RANSAC):
98
+ MIN_GEO_INLIERS = 12
99
+
100
+ # Patch matching params
101
+ TOP_K_PATCHES = 60 # use top-K best matching patches (ignores background)
102
+ SIFT_FEATURES = 2000
103
+ LOWE_RATIO = 0.75
104
+
105
+
106
+ # ═══════════════════════════════════════════════════════════════════════════
107
+ # LOGGING
108
+ # ═══════════════════════════════════════════════════════════════════════════
109
+ def log(msg, style=""):
110
+ if RICH:
111
+ console.print(f"[{style}]{msg}[/{style}]" if style else msg)
112
+ else:
113
+ print(msg)
114
+
115
+
116
+ def collect_images(folder: str) -> list[Path]:
117
+ p = Path(folder)
118
+ if not p.exists():
119
+ raise FileNotFoundError(f"Folder not found: {folder}")
120
+ imgs = sorted([f for f in p.rglob("*") if f.suffix.lower() in IMAGE_EXTS])
121
+ if not imgs:
122
+ raise ValueError(f"No images in: {folder}")
123
+ return imgs
124
+
125
+
126
+ # ═══════════════════════════════════════════════════════════════════════════
127
+ # STAGE 1 + 2 — DINOv2 EMBEDDER
128
+ # Extracts BOTH CLS (global) and patch (local) embeddings
129
+ # ═══════════════════════════════════════════════════════════════════════════
130
+ class DinoEmbedder:
131
+ """
132
+ DINOv2-Large produces:
133
+ • CLS token → 1 × 1024 vector (global semantic identity)
134
+ • Patch tokens → N × 1024 matrix (local spatial features)
135
+
136
+ Patch tokens are the KEY for robustness:
137
+ - If an object is partially cropped, only some patches appear
138
+ - We match patches individually → robust to partial views
139
+ - DINOv2 patch features are inherently robust to lighting/scale
140
+ """
141
+
142
+ def __init__(self):
143
+ log(f" Loading {MODEL_ID} on [{DEVICE.upper()}] ...", "dim")
144
+ self.proc = AutoImageProcessor.from_pretrained(MODEL_ID)
145
+ self.model = AutoModel.from_pretrained(MODEL_ID).to(DEVICE).eval()
146
+ log(" DINOv2 model ready!", "dim green")
147
+
148
+ @torch.no_grad()
149
+ def embed(self, path: Path) -> dict:
150
+ """
151
+ Returns dict with:
152
+ 'cls' : np.ndarray (1024,) L2-normalised global embedding
153
+ 'patches': np.ndarray (N, 1024) L2-normalised patch embeddings
154
+ """
155
+ img = Image.open(path).convert("RGB")
156
+ inputs = self.proc(images=img, return_tensors="pt").to(DEVICE)
157
+ out = self.model(**inputs)
158
+
159
+ hidden = out.last_hidden_state # (1, N+1, 1024)
160
+
161
+ # CLS token
162
+ cls = hidden[:, 0, :] # (1, 1024)
163
+ cls = cls / cls.norm(dim=-1, keepdim=True)
164
+ cls = cls.squeeze().cpu().numpy()
165
+
166
+ # Patch tokens (all except CLS)
167
+ patches = hidden[:, 1:, :] # (1, N, 1024)
168
+ patches = patches / patches.norm(dim=-1, keepdim=True)
169
+ patches = patches.squeeze().cpu().numpy() # (N, 1024)
170
+
171
+ return {"cls": cls, "patches": patches}
172
+
173
+ def embed_all(self, paths: list[Path], label: str) -> dict:
174
+ results = {}
175
+ if RICH:
176
+ with Progress(SpinnerColumn(), TextColumn(f"[cyan]{label}"),
177
+ BarColumn(), TaskProgressColumn(), console=console) as prog:
178
+ task = prog.add_task("", total=len(paths))
179
+ for p in paths:
180
+ try: results[p] = self.embed(p)
181
+ except Exception as e:
182
+ console.print(f" [red]skip {p.name}: {e}[/red]")
183
+ prog.advance(task)
184
+ else:
185
+ for i, p in enumerate(paths, 1):
186
+ print(f" [{i}/{len(paths)}] {p.name}")
187
+ try: results[p] = self.embed(p)
188
+ except Exception as e:
189
+ print(f" skip: {e}")
190
+ return results
191
+
192
+
193
+ # ═══════════════════════════════════════════════════════════════════════════
194
+ # PATCH-LEVEL SIMILARITY
195
+ # ═══════════════════════════════════════════════════════════════════════════
196
+ def patch_similarity(patches_a: np.ndarray, patches_b: np.ndarray) -> float:
197
+ """
198
+ For each patch in A, find its BEST matching patch in B.
199
+ Take the mean of the top-K matches (ignores background / non-matching patches).
200
+
201
+ This is robust because:
202
+ - Even if only 30% of the object is visible in one image,
203
+ those visible patches will still find their counterparts.
204
+ - Background patches will have low scores → filtered by top-K.
205
+
206
+ patches_a: (Na, D)
207
+ patches_b: (Nb, D)
208
+ Returns: float in [0, 1]
209
+ """
210
+ # Full cross-similarity matrix: (Na, Nb)
211
+ sim_matrix = patches_a @ patches_b.T # all L2-normalised → cosine
212
+
213
+ # For each patch in A: best match score in B
214
+ max_per_patch = sim_matrix.max(axis=1) # (Na,)
215
+
216
+ # Also for each patch in B: best match in A (symmetric check)
217
+ max_per_patch_b = sim_matrix.max(axis=0) # (Nb,)
218
+
219
+ # Combine: top-K from both directions → robust to asymmetric crops
220
+ all_scores = np.concatenate([max_per_patch, max_per_patch_b])
221
+ K = min(TOP_K_PATCHES * 2, len(all_scores))
222
+ top_k = np.partition(all_scores, -K)[-K:]
223
+
224
+ return float(top_k.mean())
225
+
226
+
227
+ # ═══════════════════════════════════════════════════════════════════════════
228
+ # STAGE 3 — GEOMETRIC VERIFIER (SIFT + RANSAC)
229
+ # ═══════════════════════════════════════════════════════════════════════════
230
+ class GeoVerifier:
231
+ """
232
+ Last resort when semantic evidence is uncertain.
233
+ Uses classical SIFT keypoints + Lowe ratio test + RANSAC homography.
234
+ Returns (confirmed, inlier_count).
235
+ """
236
+
237
+ def __init__(self):
238
+ self.ok = CV2_OK
239
+ if self.ok:
240
+ self.sift = cv2.SIFT_create(nfeatures=SIFT_FEATURES)
241
+ self.matcher = cv2.BFMatcher(cv2.NORM_L2, crossCheck=False)
242
+
243
+ def verify(self, a: Path, b: Path) -> tuple[bool, int]:
244
+ if not self.ok:
245
+ return False, 0
246
+ try:
247
+ ga = cv2.imread(str(a), cv2.IMREAD_GRAYSCALE)
248
+ gb = cv2.imread(str(b), cv2.IMREAD_GRAYSCALE)
249
+ if ga is None or gb is None:
250
+ return False, 0
251
+
252
+ kp_a, des_a = self.sift.detectAndCompute(ga, None)
253
+ kp_b, des_b = self.sift.detectAndCompute(gb, None)
254
+
255
+ if des_a is None or des_b is None or len(kp_a) < 6 or len(kp_b) < 6:
256
+ return False, 0
257
+
258
+ raw = self.matcher.knnMatch(des_a, des_b, k=2)
259
+ good = [m for m, n in raw if m.distance < LOWE_RATIO * n.distance]
260
+
261
+ if len(good) < 6:
262
+ return False, 0
263
+
264
+ src = np.float32([kp_a[m.queryIdx].pt for m in good]).reshape(-1, 1, 2)
265
+ dst = np.float32([kp_b[m.trainIdx].pt for m in good]).reshape(-1, 1, 2)
266
+ _, mask = cv2.findHomography(src, dst, cv2.RANSAC, 5.0)
267
+
268
+ if mask is None:
269
+ return False, 0
270
+
271
+ inliers = int(mask.sum())
272
+ return inliers >= MIN_GEO_INLIERS, inliers
273
+
274
+ except Exception:
275
+ return False, 0
276
+
277
+
278
+ # ═══════════════════════════════════════════════════════════════════════════
279
+ # RESULT
280
+ # ═══════════════════════════════════════════════════════════════════════════
281
+ class ItemResult:
282
+ def __init__(self, before_path: Path):
283
+ self.before_path = before_path
284
+ self.best_match = None # Path
285
+ self.cls_score = 0.0
286
+ self.patch_score = 0.0
287
+ self.geo_inliers = 0
288
+ self.stage_used = 0 # which stage made the decision
289
+ self.found = False
290
+
291
+ @property
292
+ def confidence(self):
293
+ s = self.cls_score
294
+ if not self.found: return "—"
295
+ if s >= 0.92: return "VERY HIGH"
296
+ if s >= 0.88: return "HIGH"
297
+ if self.patch_score >= 0.72: return "MEDIUM (patch-verified)"
298
+ if self.geo_inliers >= 12: return "MEDIUM (geo-verified)"
299
+ return "LOW"
300
+
301
+
302
+ # ═══════════════════════════════════════════════════════════════════════════
303
+ # MAIN ENGINE
304
+ # ═══════════════════════════════════════════════════════════════════════════
305
+ class InventoryMatcher:
306
+
307
+ def __init__(self):
308
+ self.emb = DinoEmbedder()
309
+ self.geo = GeoVerifier()
310
+
311
+ # ── run ───────────────────────────────────────────────────────────────
312
+ def run(self, before_folder: str, after_folder: str) -> list[ItemResult]:
313
+
314
+ before_paths = collect_images(before_folder)
315
+ after_paths = collect_images(after_folder)
316
+
317
+ if RICH:
318
+ console.print(Panel(
319
+ f"[bold]📁 Before:[/bold] {before_folder} "
320
+ f"([yellow]{len(before_paths)} items[/yellow])\n"
321
+ f"[bold]📁 After: [/bold] {after_folder} "
322
+ f"([yellow]{len(after_paths)} images[/yellow])\n"
323
+ f"[bold]⚙️ Stages:[/bold] DINOv2-CLS → Patch matching → SIFT+RANSAC\n"
324
+ f"[bold]🖥️ Device:[/bold] {DEVICE.upper()}",
325
+ title="[bold cyan]Furniture Inventory Matcher[/bold cyan]",
326
+ border_style="cyan"
327
+ ))
328
+ else:
329
+ print(f"\n{'='*60}")
330
+ print(f"Before: {before_folder} ({len(before_paths)} items)")
331
+ print(f"After: {after_folder} ({len(after_paths)} images)")
332
+ print(f"Stages: DINOv2-CLS → Patch → SIFT+RANSAC")
333
+ print(f"Device: {DEVICE.upper()}\n{'='*60}\n")
334
+
335
+ # ── Embed all ──────────────────────────────────────────────────────
336
+ log("\n[Step 1/3] Embedding BEFORE folder …")
337
+ before_embs = self.emb.embed_all(before_paths, "BEFORE ")
338
+
339
+ log("\n[Step 2/3] Embedding AFTER folder …")
340
+ after_embs = self.emb.embed_all(after_paths, "AFTER ")
341
+
342
+ # ── Pre-compute CLS similarity matrix ─────────────────────────────
343
+ log("\n[Step 3/3] Matching items …")
344
+ before_list = list(before_embs.keys())
345
+ after_list = list(after_embs.keys())
346
+
347
+ B_cls = np.stack([before_embs[p]["cls"] for p in before_list]) # (B, D)
348
+ A_cls = np.stack([after_embs[p]["cls"] for p in after_list]) # (A, D)
349
+ cls_sim = B_cls @ A_cls.T # (B, A)
350
+
351
+ results = []
352
+
353
+ for b_idx, b_path in enumerate(before_list):
354
+ item = ItemResult(b_path)
355
+ row = cls_sim[b_idx] # CLS similarities to all after images
356
+
357
+ # ── Find best candidate in AFTER ──────────────────────────────
358
+ best_a_idx = int(np.argmax(row))
359
+ best_score = float(row[best_a_idx])
360
+ best_a_path = after_list[best_a_idx]
361
+
362
+ item.cls_score = best_score
363
+ item.best_match = best_a_path
364
+
365
+ # ══════════════════════════════════════════════════════════════
366
+ # STAGE 1: CLS global score
367
+ # ══════════════════════════════════════════════════════════════
368
+ if best_score >= CLS_CONFIRM_FOUND:
369
+ item.found = True
370
+ item.stage_used = 1
371
+
372
+ elif best_score < CLS_CONFIRM_MISSING:
373
+ item.found = False
374
+ item.stage_used = 1
375
+
376
+ else:
377
+ # ══════════════════════════════════════════════════════════
378
+ # STAGE 2: PATCH-level matching
379
+ # Uncertain zone → look at individual patch correspondences
380
+ # ══════════════════════════════════════════════════════════
381
+ patches_b = before_embs[b_path]["patches"]
382
+ patches_a = after_embs[best_a_path]["patches"]
383
+ p_score = patch_similarity(patches_b, patches_a)
384
+ item.patch_score = p_score
385
+ item.stage_used = 2
386
+
387
+ if p_score >= PATCH_CONFIRM_FOUND:
388
+ item.found = True
389
+
390
+ elif p_score < PATCH_CONFIRM_MISSING:
391
+ item.found = False
392
+
393
+ else:
394
+ # ══════════════════════════════════════════════════════
395
+ # STAGE 3: GEOMETRIC verification (SIFT + RANSAC)
396
+ # Last resort — pixel-level structural proof
397
+ # ══════════════════════════════════════════════════════
398
+ confirmed, inliers = self.geo.verify(b_path, best_a_path)
399
+ item.geo_inliers = inliers
400
+ item.found = confirmed
401
+ item.stage_used = 3
402
+
403
+ results.append(item)
404
+
405
+ return results
406
+
407
+ # ── Console report ─────────────────────────────────────────────────────
408
+ def print_report(self, results: list[ItemResult]):
409
+ found = [r for r in results if r.found]
410
+ missing = [r for r in results if not r.found]
411
+
412
+ if RICH:
413
+ console.print(Panel(
414
+ f"[bold green]✅ FOUND: {len(found)} / {len(results)}[/bold green]\n"
415
+ f"[bold red]❌ MISSING: {len(missing)} / {len(results)}[/bold red]",
416
+ title="[bold]📋 Inventory Report[/bold]", border_style="magenta"
417
+ ))
418
+
419
+ if found:
420
+ t = Table(title="[green]✅ FOUND[/green]",
421
+ border_style="green", header_style="bold")
422
+ t.add_column("Before", style="cyan", no_wrap=True)
423
+ t.add_column("After", style="yellow", no_wrap=True)
424
+ t.add_column("CLS", width=6)
425
+ t.add_column("Patch", width=6)
426
+ t.add_column("Geo", width=5)
427
+ t.add_column("Stage", width=5)
428
+ t.add_column("Confidence",width=22)
429
+ for r in found:
430
+ ps = f"{r.patch_score:.3f}" if r.patch_score else "—"
431
+ geo = str(r.geo_inliers) if r.geo_inliers else "—"
432
+ t.add_row(r.before_path.name, r.best_match.name,
433
+ f"{r.cls_score:.3f}", ps, geo,
434
+ str(r.stage_used), r.confidence)
435
+ console.print(t)
436
+
437
+ if missing:
438
+ t = Table(title="[red]❌ MISSING[/red]",
439
+ border_style="red", header_style="bold")
440
+ t.add_column("Before", style="cyan", no_wrap=True)
441
+ t.add_column("Best CLS", width=8)
442
+ t.add_column("Best Patch", width=9)
443
+ t.add_column("Closest In", style="dim", no_wrap=True)
444
+ for r in missing:
445
+ ps = f"{r.patch_score:.3f}" if r.patch_score else "—"
446
+ nm = r.best_match.name if r.best_match else "—"
447
+ t.add_row(r.before_path.name, f"{r.cls_score:.3f}", ps, nm)
448
+ console.print(t)
449
+ else:
450
+ print(f"\nFOUND: {len(found)}/{len(results)}")
451
+ print(f"MISSING: {len(missing)}/{len(results)}")
452
+ if found:
453
+ print("\n✅ FOUND:")
454
+ for r in found:
455
+ print(f" {r.before_path.name:<35} → {r.best_match.name} "
456
+ f"(cls:{r.cls_score:.3f} patch:{r.patch_score:.3f})")
457
+ if missing:
458
+ print("\n❌ MISSING:")
459
+ for r in missing:
460
+ print(f" {r.before_path.name:<35} best cls:{r.cls_score:.3f}")
461
+
462
+ # ── HTML report ────────────────────────────────────────────────────────
463
+ def save_html(self, results: list[ItemResult], out_dir: str) -> str:
464
+ out = Path(out_dir)
465
+ out.mkdir(parents=True, exist_ok=True)
466
+
467
+ def b64img(path: Path):
468
+ try:
469
+ data = base64.b64encode(path.read_bytes()).decode()
470
+ ext = path.suffix.lstrip(".") or "jpeg"
471
+ return f"data:image/{ext};base64,{data}"
472
+ except Exception:
473
+ return ""
474
+
475
+ cards = ""
476
+ for r in results:
477
+ is_found = r.found
478
+ color = "#16a34a" if is_found else "#dc2626"
479
+ bg_color = "#0f2d1a" if is_found else "#2d0f0f"
480
+ border_c = "#22c55e" if is_found else "#ef4444"
481
+ icon = "✅" if is_found else "❌"
482
+ label = "FOUND" if is_found else "MISSING"
483
+
484
+ before_src = b64img(r.before_path)
485
+ after_html = ""
486
+ if r.best_match:
487
+ after_src = b64img(r.best_match)
488
+ after_name = r.best_match.name
489
+ after_html = f"""
490
+ <div class="arrow">→</div>
491
+ <div class="img-box">
492
+ <div class="box-label">Best match · AFTER</div>
493
+ <img src="{after_src}" alt="{after_name}"/>
494
+ <div class="fname">{after_name}</div>
495
+ </div>"""
496
+
497
+ stage_labels = {1: "CLS only", 2: "Patch match", 3: "SIFT+RANSAC"}
498
+ stage_str = stage_labels.get(r.stage_used, "")
499
+
500
+ scores_html = f"""
501
+ <div class="scores">
502
+ <span>CLS: <b>{r.cls_score:.3f}</b></span>
503
+ {'<span>Patch: <b>' + f'{r.patch_score:.3f}' + '</b></span>' if r.patch_score else ''}
504
+ {'<span>Geo inliers: <b>' + str(r.geo_inliers) + '</b></span>' if r.geo_inliers else ''}
505
+ <span>Decision: <b>{stage_str}</b></span>
506
+ {'<span>Confidence: <b>' + r.confidence + '</b></span>' if is_found else ''}
507
+ </div>"""
508
+
509
+ cards += f"""
510
+ <div class="card" style="background:{bg_color};border-color:{border_c}">
511
+ <div class="badge" style="background:{color}">{icon} {label}</div>
512
+ <div class="img-row">
513
+ <div class="img-box">
514
+ <div class="box-label">BEFORE</div>
515
+ <img src="{before_src}" alt="{r.before_path.name}"/>
516
+ <div class="fname">{r.before_path.name}</div>
517
+ </div>
518
+ {after_html}
519
+ </div>
520
+ {scores_html}
521
+ </div>"""
522
+
523
+ n_found = sum(1 for r in results if r.found)
524
+ n_missing = len(results) - n_found
525
+
526
+ html = f"""<!DOCTYPE html>
527
+ <html lang="en">
528
+ <head>
529
+ <meta charset="UTF-8">
530
+ <meta name="viewport" content="width=device-width, initial-scale=1.0">
531
+ <title>Furniture Inventory Report</title>
532
+ <style>
533
+ *{{box-sizing:border-box;margin:0;padding:0;}}
534
+ body{{font-family:'Segoe UI',sans-serif;background:#0a0f1e;color:#e2e8f0;padding:2rem;}}
535
+ h1{{text-align:center;font-size:2rem;margin-bottom:.3rem;
536
+ background:linear-gradient(135deg,#38bdf8,#818cf8);
537
+ -webkit-background-clip:text;-webkit-text-fill-color:transparent;}}
538
+ .sub{{text-align:center;color:#64748b;font-size:.85rem;margin-bottom:1.5rem;}}
539
+
540
+ .summary{{display:flex;justify-content:center;gap:1.5rem;margin:1.5rem 0;flex-wrap:wrap;}}
541
+ .stat{{background:#1e293b;border-radius:14px;padding:.8rem 2rem;
542
+ text-align:center;border:1px solid #334155;}}
543
+ .stat .num{{font-size:2.2rem;font-weight:800;}}
544
+ .stat .lbl{{font-size:.8rem;color:#94a3b8;margin-top:.1rem;}}
545
+
546
+ .filters{{text-align:center;margin-bottom:1.5rem;}}
547
+ .fbtn{{background:#1e293b;border:1px solid #334155;color:#e2e8f0;
548
+ padding:.35rem 1.1rem;border-radius:999px;cursor:pointer;
549
+ margin:0 .25rem;font-size:.8rem;transition:.15s;}}
550
+ .fbtn:hover,.fbtn.active{{background:#38bdf8;color:#0a0f1e;border-color:#38bdf8;}}
551
+
552
+ .card{{border-radius:16px;padding:1.4rem;margin-bottom:1.4rem;
553
+ border:1px solid;box-shadow:0 4px 20px rgba(0,0,0,.5);}}
554
+ .badge{{display:inline-block;padding:.25rem .9rem;border-radius:999px;
555
+ font-weight:700;color:#fff;font-size:.82rem;margin-bottom:1rem;}}
556
+ .img-row{{display:flex;align-items:center;gap:1.2rem;flex-wrap:wrap;}}
557
+ .img-box{{flex:1;min-width:160px;text-align:center;}}
558
+ .img-box img{{max-width:100%;max-height:250px;border-radius:10px;
559
+ object-fit:contain;border:2px solid #334155;}}
560
+ .box-label{{font-size:.65rem;color:#94a3b8;text-transform:uppercase;
561
+ letter-spacing:.08em;margin-bottom:.3rem;}}
562
+ .fname{{font-size:.72rem;color:#475569;margin-top:.3rem;word-break:break-all;}}
563
+ .arrow{{font-size:2rem;color:#38bdf8;font-weight:bold;flex-shrink:0;}}
564
+ .scores{{display:flex;flex-wrap:wrap;gap:.6rem;margin-top:.9rem;}}
565
+ .scores span{{background:#0f172a;padding:.2rem .6rem;border-radius:6px;
566
+ font-size:.75rem;color:#94a3b8;}}
567
+ .scores b{{color:#e2e8f0;}}
568
+ </style>
569
+ </head>
570
+ <body>
571
+ <h1>🛋️ Furniture Inventory Report</h1>
572
+ <p class="sub">DINOv2-Large · Patch Matching · SIFT+RANSAC · {len(results)} items</p>
573
+
574
+ <div class="summary">
575
+ <div class="stat"><div class="num" style="color:#22c55e">{n_found}</div>
576
+ <div class="lbl">✅ FOUND</div></div>
577
+ <div class="stat"><div class="num" style="color:#ef4444">{n_missing}</div>
578
+ <div class="lbl">❌ MISSING</div></div>
579
+ <div class="stat"><div class="num" style="color:#94a3b8">{len(results)}</div>
580
+ <div class="lbl">Total Items</div></div>
581
+ </div>
582
+
583
+ <div class="filters">
584
+ <button class="fbtn active" onclick="filt('all',this)">All</button>
585
+ <button class="fbtn" onclick="filt('found',this)">✅ Found</button>
586
+ <button class="fbtn" onclick="filt('missing',this)">❌ Missing</button>
587
+ </div>
588
+
589
+ <div id="cards">{cards}</div>
590
+
591
+ <script>
592
+ function filt(type,btn){{
593
+ document.querySelectorAll('.fbtn').forEach(b=>b.classList.remove('active'));
594
+ btn.classList.add('active');
595
+ document.querySelectorAll('.card').forEach(c=>{{
596
+ const isMissing=c.querySelector('.badge').textContent.includes('MISSING');
597
+ if(type==='all') c.style.display='';
598
+ else if(type==='found') c.style.display=isMissing?'none':'';
599
+ else c.style.display=isMissing?'':'none';
600
+ }});
601
+ }}
602
+ </script>
603
+ </body>
604
+ </html>"""
605
+
606
+ path = out / "inventory_report.html"
607
+ path.write_text(html, encoding="utf-8")
608
+ return str(path)
609
+
610
+
611
+ # ═══════════════════════════════════════════════════════════════════════════
612
+ # CLI
613
+ # ═══════════════════════════════════════════════════════════════════════════
614
+ def main():
615
+ p = argparse.ArgumentParser(
616
+ description="Furniture Inventory: find FOUND / MISSING items between two folders."
617
+ )
618
+ p.add_argument("--before", required=True, help="Folder with reference images (before)")
619
+ p.add_argument("--after", required=True, help="Folder with images to compare (after)")
620
+ p.add_argument("--output", default=None, help="Output folder for HTML report")
621
+ p.add_argument("--strict", action="store_true",
622
+ help="Raise thresholds to reduce false positives even further")
623
+ args = p.parse_args()
624
+
625
+ if args.strict:
626
+ global CLS_CONFIRM_FOUND, PATCH_CONFIRM_FOUND, MIN_GEO_INLIERS
627
+ CLS_CONFIRM_FOUND = 0.90
628
+ PATCH_CONFIRM_FOUND = 0.76
629
+ MIN_GEO_INLIERS = 18
630
+ log("Strict mode ON — thresholds raised ↑", "yellow")
631
+
632
+ t0 = time.time()
633
+ engine = InventoryMatcher()
634
+ results = engine.run(args.before, args.after)
635
+ engine.print_report(results)
636
+
637
+ if args.output:
638
+ rpt = engine.save_html(results, args.output)
639
+ log(f"\n📄 Report → {rpt}", "bold green")
640
+
641
+ log(f"\n⏱ Finished in {time.time()-t0:.1f}s", "dim")
642
+
643
+
644
+ if __name__ == "__main__":
645
+ main()
mllm_inspector.py ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import base64
3
+ import json
4
+ import google.generativeai as genai
5
+ from groq import Groq
6
+ from openai import OpenAI
7
+
8
+ def encode_image(image_path):
9
+ with open(image_path, "rb") as image_file:
10
+ return base64.b64encode(image_file.read()).decode('utf-8')
11
+
12
+ def get_prompt():
13
+ return """
14
+ You are an expert property inspector.
15
+ Compare these two images: the first one is the "Before" state, and the second one is the "After" state of the same object.
16
+ Is there any damage, breakage, deep scratch, or negative change in the "After" image?
17
+
18
+ Respond STRICTLY in JSON format with three keys:
19
+ 1. "description": A detailed English description of the damage found. If no damage is found, state that it is intact.
20
+ 2. "target_phrase": A short English phrase (2-4 words) describing the overall damaged parts (e.g., "damaged table and chairs"). If no damage, return "None".
21
+ 3. "target_phrases_list": A list of highly specific, short English phrases (1-3 words each), breaking down every single damaged object separately to be used individually by an object detection model. (e.g., ["torn chair", "broken table corner", "cracked wall", "peeling paint", "torn curtain"]). If no damage, return [].
22
+
23
+ Do not output any markdown text outside the JSON object.
24
+ """
25
+
26
+ def inspect_with_gemini(image_before_path, image_after_path, api_key):
27
+ try:
28
+ from PIL import Image
29
+ genai.configure(api_key=api_key)
30
+ model = genai.GenerativeModel('gemini-2.0-flash')
31
+ img_before = Image.open(image_before_path)
32
+ img_after = Image.open(image_after_path)
33
+
34
+ response = model.generate_content([get_prompt(), img_before, img_after])
35
+ text_response = response.text
36
+ if "```json" in text_response:
37
+ text_response = text_response.split("```json")[1].split("```")[0].strip()
38
+ return json.loads(text_response)
39
+ except Exception as e:
40
+ print(f"\n❌ خطأ في Gemini: {e}")
41
+ return None
42
+
43
+ def inspect_with_groq(image_before_path, image_after_path, api_key):
44
+ try:
45
+ client = Groq(api_key=api_key)
46
+ base64_before = encode_image(image_before_path)
47
+ base64_after = encode_image(image_after_path)
48
+
49
+ response = client.chat.completions.create(
50
+ model="llama-3.2-11b-vision-preview",
51
+ messages=[
52
+ {
53
+ "role": "user",
54
+ "content": [
55
+ {"type": "text", "text": get_prompt()},
56
+ {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_before}"}},
57
+ {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_after}"}}
58
+ ]
59
+ }
60
+ ],
61
+ temperature=0.1
62
+ )
63
+
64
+ text_response = response.choices[0].message.content
65
+ import re
66
+ json_match = re.search(r'\{.*\}', text_response, re.DOTALL)
67
+ if json_match:
68
+ text_response = json_match.group(0)
69
+
70
+ return json.loads(text_response)
71
+ except Exception as e:
72
+ print(f"\n❌ خطأ في Groq: {e}")
73
+ raise e
74
+
75
+ def inspect_with_openai(image_before_path, image_after_path, api_key):
76
+ try:
77
+ client = OpenAI(api_key=api_key)
78
+ base64_before = encode_image(image_before_path)
79
+ base64_after = encode_image(image_after_path)
80
+
81
+ response = client.chat.completions.create(
82
+ model="gpt-4o-mini",
83
+ messages=[
84
+ {
85
+ "role": "user",
86
+ "content": [
87
+ {"type": "text", "text": get_prompt()},
88
+ {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_before}"}},
89
+ {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_after}"}}
90
+ ]
91
+ }
92
+ ],
93
+ temperature=0.1,
94
+ response_format={"type": "json_object"}
95
+ )
96
+ return json.loads(response.choices[0].message.content)
97
+ except Exception as e:
98
+ print(f"\n❌ خطأ في OpenAI: {e}")
99
+ return None
100
+
101
+ def inspect_with_openrouter(image_before_path, image_after_path, api_key):
102
+ try:
103
+ # OpenRouter uses the exact same OpenAI library but a different URL
104
+ client = OpenAI(
105
+ base_url="https://openrouter.ai/api/v1",
106
+ api_key=api_key,
107
+ )
108
+ base64_before = encode_image(image_before_path)
109
+ base64_after = encode_image(image_after_path)
110
+
111
+ # Free vision model on OpenRouter!
112
+ response = client.chat.completions.create(
113
+ model="nvidia/nemotron-nano-12b-v2-vl:free",
114
+ messages=[
115
+ {
116
+ "role": "user",
117
+ "content": [
118
+ {"type": "text", "text": get_prompt()},
119
+ {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_before}"}},
120
+ {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_after}"}}
121
+ ]
122
+ }
123
+ ],
124
+ temperature=0.1
125
+ )
126
+
127
+ if not hasattr(response, 'choices') or not response.choices:
128
+ raise ValueError(f"OpenRouter returned an invalid response: {response}")
129
+
130
+ text_response = response.choices[0].message.content
131
+ import re
132
+ json_match = re.search(r'\{.*\}', text_response, re.DOTALL)
133
+ if json_match:
134
+ text_response = json_match.group(0)
135
+
136
+ return json.loads(text_response)
137
+ except Exception as e:
138
+ print(f"\n❌ خطأ في الاتصال: {e}")
139
+ raise e
report_generator.py ADDED
@@ -0,0 +1,507 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import base64
3
+ from pathlib import Path
4
+
5
+ def generate_combined_report(results, out_dir):
6
+ """
7
+ results is a list of dicts, each containing:
8
+ - 'before_path': Path
9
+ - 'after_path': Path or None
10
+ - 'status': 'INTACT' | 'DAMAGED' | 'MISSING'
11
+ - 'cls_score': float
12
+ - 'patch_score': float
13
+ - 'geo_inliers': int
14
+ - 'confidence': str
15
+ - 'stage_used': int
16
+ - 'damage_description': str
17
+ - 'target_phrases_list': list of str
18
+ - 'masked_after_path': Path or None
19
+ """
20
+ out = Path(out_dir)
21
+ out.mkdir(parents=True, exist_ok=True)
22
+
23
+ def b64img(path):
24
+ if not path or not os.path.exists(path):
25
+ return ""
26
+ try:
27
+ p = Path(path)
28
+ data = base64.b64encode(p.read_bytes()).decode()
29
+ ext = p.suffix.lstrip(".") or "jpeg"
30
+ return f"data:image/{ext};base64,{data}"
31
+ except Exception as e:
32
+ print(f"Error encoding image {path}: {e}")
33
+ return ""
34
+
35
+ cards = ""
36
+ for r in results:
37
+ status = r['status']
38
+
39
+ # Determine styling based on status
40
+ if status == 'INTACT':
41
+ color = "#10b981" # Emerald
42
+ bg_color = "rgba(16, 185, 129, 0.05)"
43
+ border_c = "rgba(16, 185, 129, 0.2)"
44
+ icon = "✅"
45
+ label = "INTACT / سليم"
46
+ elif status == 'DAMAGED':
47
+ color = "#f59e0b" # Amber
48
+ bg_color = "rgba(245, 158, 11, 0.05)"
49
+ border_c = "rgba(245, 158, 11, 0.25)"
50
+ icon = "⚠️"
51
+ label = "DAMAGED / تالف"
52
+ else:
53
+ color = "#ef4444" # Red
54
+ bg_color = "rgba(239, 68, 68, 0.05)"
55
+ border_c = "rgba(239, 68, 68, 0.2)"
56
+ icon = "❌"
57
+ label = "MISSING / مفقود"
58
+
59
+ before_src = b64img(r['before_path'])
60
+
61
+ after_html = ""
62
+ if r['after_path']:
63
+ after_src = b64img(r['after_path'])
64
+ after_name = Path(r['after_path']).name
65
+ after_html = f"""
66
+ <div class="arrow">→</div>
67
+ <div class="img-box">
68
+ <div class="box-label">AFTER IMAGE / الصورة بعد</div>
69
+ <img src="{after_src}" alt="{after_name}"/>
70
+ <div class="fname">{after_name}</div>
71
+ </div>"""
72
+
73
+ mask_html = ""
74
+ if status == 'DAMAGED' and r['masked_after_path']:
75
+ mask_src = b64img(r['masked_after_path'])
76
+ mask_name = Path(r['masked_after_path']).name
77
+ mask_html = f"""
78
+ <div class="arrow">→</div>
79
+ <div class="img-box highlighted">
80
+ <div class="box-label" style="color:#f59e0b">🎯 DETECTED DAMAGE (SAM MASK) / الضرر المكتشف (قناع SAM)</div>
81
+ <img src="{mask_src}" alt="{mask_name}" style="border-color:#f59e0b"/>
82
+ <div class="fname" style="color:#f59e0b">{mask_name}</div>
83
+ </div>"""
84
+
85
+ stage_labels = {1: "DINOv2 CLS", 2: "DINOv2 Patch", 3: "SIFT+RANSAC"}
86
+ stage_str = stage_labels.get(r.get('stage_used', 0), "None")
87
+
88
+ # Matching metrics block
89
+ scores_html = ""
90
+ if r['after_path']:
91
+ scores_html = f"""
92
+ <div class="metric-chip">Match / مطابقة: <b>{stage_str}</b></div>
93
+ <div class="metric-chip">CLS score / درجة CLS: <b>{r['cls_score']:.3f}</b></div>
94
+ """
95
+ if r['patch_score']:
96
+ scores_html += f'<div class="metric-chip">Patch score / درجة الرقعة: <b>{r["patch_score"]:.3f}</b></div>'
97
+ if r['geo_inliers']:
98
+ scores_html += f'<div class="metric-chip">Geo Inliers / مطابقة هندسية: <b>{r["geo_inliers"]}</b></div>'
99
+ if r.get('confidence'):
100
+ scores_html += f'<div class="metric-chip">Confidence / ثقة: <b>{r["confidence"]}</b></div>'
101
+
102
+ # Damage Report Text Block
103
+ report_details_html = ""
104
+ if status == 'DAMAGED':
105
+ phrases_badges = "".join([f'<span class="phrase-tag">{p}</span>' for p in r['target_phrases_list']])
106
+ report_details_html = f"""
107
+ <div class="damage-report-box">
108
+ <h4>📝 AI Damage Assessment Report / تقرير فحص التلفيات بالذكاء الاصطناعي</h4>
109
+ <p class="desc">{r['damage_description']}</p>
110
+ <div class="phrases-container">
111
+ <strong>Segmented features / الأجزاء المحددة:</strong>
112
+ <div class="phrase-tags-row">{phrases_badges}</div>
113
+ </div>
114
+ </div>
115
+ """
116
+ elif status == 'INTACT':
117
+ report_details_html = f"""
118
+ <div class="intact-report-box">
119
+ <p>✨ AI analysis confirms this item is intact. No major damage or negative changes were detected. / يؤكد تحليل الذكاء الاصطناعي أن هذا العنصر سليم، ولم يتم رصد أي تلفيات أو تغييرات سلبية.</p>
120
+ </div>
121
+ """
122
+ else:
123
+ report_details_html = f"""
124
+ <div class="missing-report-box">
125
+ <p>🔍 This item from the before inventory could not be matched with any item in the after inventory. It may have been moved, stolen, or removed. / لم يتم العثور على مطابقة لهذا العنصر في صور البعد، قد يكون تم نقله، سرقته أو إزالته.</p>
126
+ </div>
127
+ """
128
+
129
+ cards += f"""
130
+ <div class="card" data-status="{status}" style="background:{bg_color};border-color:{border_c}">
131
+ <div class="card-header">
132
+ <div class="badge" style="background:{color}">{icon} {label}</div>
133
+ <div class="metrics-row">{scores_html}</div>
134
+ </div>
135
+ <div class="img-row">
136
+ <div class="img-box">
137
+ <div class="box-label">BEFORE IMAGE / الصورة قبل</div>
138
+ <img src="{before_src}" alt="{Path(r['before_path']).name}"/>
139
+ <div class="fname">{Path(r['before_path']).name}</div>
140
+ </div>
141
+ {after_html}
142
+ {mask_html}
143
+ </div>
144
+ {report_details_html}
145
+ </div>"""
146
+
147
+ n_total = len(results)
148
+ n_intact = sum(1 for r in results if r['status'] == 'INTACT')
149
+ n_damaged = sum(1 for r in results if r['status'] == 'DAMAGED')
150
+ n_missing = sum(1 for r in results if r['status'] == 'MISSING')
151
+
152
+ html = f"""<!DOCTYPE html>
153
+ <html lang="en">
154
+ <head>
155
+ <meta charset="UTF-8">
156
+ <meta name="viewport" content="width=device-width, initial-scale=1.0">
157
+ <title>AI Inventory & Damage Report</title>
158
+ <style>
159
+ @import url('https://fonts.googleapis.com/css2?family=Inter:wght@300;400;500;600;700;800&family=Cairo:wght@300;400;600;700;800&display=swap');
160
+
161
+ * {{
162
+ box-sizing: border-box;
163
+ margin: 0;
164
+ padding: 0;
165
+ }}
166
+
167
+ body {{
168
+ font-family: 'Inter', 'Cairo', -apple-system, sans-serif;
169
+ background: #0b0f19;
170
+ color: #f1f5f9;
171
+ padding: 2.5rem 1.5rem;
172
+ line-height: 1.5;
173
+ }}
174
+
175
+ .container {{
176
+ max-width: 1200px;
177
+ margin: 0 auto;
178
+ }}
179
+
180
+ header {{
181
+ text-align: center;
182
+ margin-bottom: 2.5rem;
183
+ }}
184
+
185
+ h1 {{
186
+ font-size: 2.5rem;
187
+ font-weight: 800;
188
+ margin-bottom: 0.5rem;
189
+ letter-spacing: -0.025em;
190
+ background: linear-gradient(135deg, #60a5fa, #34d399);
191
+ -webkit-background-clip: text;
192
+ -webkit-text-fill-color: transparent;
193
+ }}
194
+
195
+ .sub {{
196
+ color: #94a3b8;
197
+ font-size: 1rem;
198
+ font-weight: 400;
199
+ }}
200
+
201
+ .summary {{
202
+ display: grid;
203
+ grid-template-columns: repeat(auto-fit, minmax(180px, 1fr));
204
+ gap: 1.25rem;
205
+ margin: 2rem 0;
206
+ }}
207
+
208
+ .stat {{
209
+ background: #1e293b;
210
+ border-radius: 16px;
211
+ padding: 1.25rem;
212
+ text-align: center;
213
+ border: 1px solid #334155;
214
+ transition: transform 0.2s, box-shadow 0.2s;
215
+ }}
216
+
217
+ .stat:hover {{
218
+ transform: translateY(-2px);
219
+ box-shadow: 0 10px 15px -3px rgba(0, 0, 0, 0.3);
220
+ }}
221
+
222
+ .stat .num {{
223
+ font-size: 2.25rem;
224
+ font-weight: 800;
225
+ line-height: 1.2;
226
+ }}
227
+
228
+ .stat .lbl {{
229
+ font-size: 0.85rem;
230
+ color: #94a3b8;
231
+ font-weight: 500;
232
+ margin-top: 0.25rem;
233
+ text-transform: uppercase;
234
+ letter-spacing: 0.05em;
235
+ }}
236
+
237
+ .filters {{
238
+ text-align: center;
239
+ margin-bottom: 2rem;
240
+ }}
241
+
242
+ .fbtn {{
243
+ background: #1e293b;
244
+ border: 1px solid #334155;
245
+ color: #cbd5e1;
246
+ padding: 0.5rem 1.5rem;
247
+ border-radius: 9999px;
248
+ cursor: pointer;
249
+ margin: 0.25rem;
250
+ font-size: 0.875rem;
251
+ font-weight: 600;
252
+ transition: all 0.2s;
253
+ }}
254
+
255
+ .fbtn:hover, .fbtn.active {{
256
+ background: #3b82f6;
257
+ color: #ffffff;
258
+ border-color: #3b82f6;
259
+ box-shadow: 0 4px 6px -1px rgba(59, 130, 246, 0.4);
260
+ }}
261
+
262
+ .card {{
263
+ border-radius: 20px;
264
+ padding: 1.75rem;
265
+ margin-bottom: 2rem;
266
+ border: 1px solid;
267
+ box-shadow: 0 10px 25px -5px rgba(0, 0, 0, 0.4);
268
+ background: #111827;
269
+ transition: transform 0.2s;
270
+ }}
271
+
272
+ .card:hover {{
273
+ transform: scale(1.005);
274
+ }}
275
+
276
+ .card-header {{
277
+ display: flex;
278
+ justify-content: space-between;
279
+ align-items: center;
280
+ flex-wrap: wrap;
281
+ gap: 1rem;
282
+ margin-bottom: 1.25rem;
283
+ border-bottom: 1px solid rgba(255, 255, 255, 0.05);
284
+ padding-bottom: 1rem;
285
+ }}
286
+
287
+ .badge {{
288
+ display: inline-block;
289
+ padding: 0.35rem 1.1rem;
290
+ border-radius: 9999px;
291
+ font-weight: 700;
292
+ color: #ffffff;
293
+ font-size: 0.85rem;
294
+ text-transform: uppercase;
295
+ letter-spacing: 0.025em;
296
+ }}
297
+
298
+ .metrics-row {{
299
+ display: flex;
300
+ flex-wrap: wrap;
301
+ gap: 0.5rem;
302
+ }}
303
+
304
+ .metric-chip {{
305
+ background: #1e293b;
306
+ border: 1px solid #334155;
307
+ padding: 0.25rem 0.75rem;
308
+ border-radius: 8px;
309
+ font-size: 0.75rem;
310
+ color: #94a3b8;
311
+ }}
312
+
313
+ .metric-chip b {{
314
+ color: #f1f5f9;
315
+ }}
316
+
317
+ .img-row {{
318
+ display: flex;
319
+ align-items: center;
320
+ gap: 1.5rem;
321
+ flex-wrap: wrap;
322
+ margin-bottom: 1.5rem;
323
+ }}
324
+
325
+ .img-box {{
326
+ flex: 1;
327
+ min-width: 260px;
328
+ text-align: center;
329
+ background: #0f172a;
330
+ padding: 1rem;
331
+ border-radius: 12px;
332
+ border: 1px solid #1e293b;
333
+ }}
334
+
335
+ .img-box img {{
336
+ max-width: 100%;
337
+ max-height: 280px;
338
+ border-radius: 8px;
339
+ object-fit: contain;
340
+ border: 2px solid #334155;
341
+ margin-top: 0.5rem;
342
+ background: #0b0f19;
343
+ }}
344
+
345
+ .box-label {{
346
+ font-size: 0.75rem;
347
+ font-weight: 700;
348
+ color: #64748b;
349
+ text-transform: uppercase;
350
+ letter-spacing: 0.08em;
351
+ }}
352
+
353
+ .fname {{
354
+ font-size: 0.75rem;
355
+ color: #475569;
356
+ margin-top: 0.5rem;
357
+ word-break: break-all;
358
+ font-family: monospace;
359
+ }}
360
+
361
+ .arrow {{
362
+ font-size: 2rem;
363
+ color: #475569;
364
+ font-weight: 800;
365
+ flex-shrink: 0;
366
+ }}
367
+
368
+ .damage-report-box {{
369
+ background: rgba(245, 158, 11, 0.03);
370
+ border: 1px dashed rgba(245, 158, 11, 0.2);
371
+ border-radius: 12px;
372
+ padding: 1.25rem;
373
+ margin-top: 1rem;
374
+ }}
375
+
376
+ .damage-report-box h4 {{
377
+ color: #f59e0b;
378
+ font-size: 0.95rem;
379
+ font-weight: 700;
380
+ margin-bottom: 0.5rem;
381
+ }}
382
+
383
+ .damage-report-box .desc {{
384
+ font-size: 0.9rem;
385
+ color: #e2e8f0;
386
+ margin-bottom: 0.75rem;
387
+ }}
388
+
389
+ .phrases-container {{
390
+ display: flex;
391
+ align-items: center;
392
+ gap: 0.75rem;
393
+ flex-wrap: wrap;
394
+ font-size: 0.8rem;
395
+ color: #94a3b8;
396
+ }}
397
+
398
+ .phrase-tags-row {{
399
+ display: flex;
400
+ flex-wrap: wrap;
401
+ gap: 0.35rem;
402
+ }}
403
+
404
+ .phrase-tag {{
405
+ background: rgba(245, 158, 11, 0.1);
406
+ color: #f59e0b;
407
+ border: 1px solid rgba(245, 158, 11, 0.2);
408
+ padding: 0.15rem 0.5rem;
409
+ border-radius: 4px;
410
+ font-weight: 600;
411
+ font-size: 0.75rem;
412
+ }}
413
+
414
+ .intact-report-box {{
415
+ background: rgba(16, 185, 129, 0.03);
416
+ border: 1px solid rgba(16, 185, 129, 0.15);
417
+ border-radius: 12px;
418
+ padding: 1rem 1.25rem;
419
+ font-size: 0.9rem;
420
+ color: #a7f3d0;
421
+ }}
422
+
423
+ .missing-report-box {{
424
+ background: rgba(239, 68, 68, 0.03);
425
+ border: 1px solid rgba(239, 68, 68, 0.15);
426
+ border-radius: 12px;
427
+ padding: 1rem 1.25rem;
428
+ font-size: 0.9rem;
429
+ color: #fca5a5;
430
+ }}
431
+
432
+ @media (max-width: 768px) {{
433
+ .arrow {{
434
+ display: none;
435
+ }}
436
+ .img-row {{
437
+ flex-direction: column;
438
+ }}
439
+ .img-box {{
440
+ width: 100%;
441
+ }}
442
+ .card-header {{
443
+ flex-direction: column;
444
+ align-items: flex-start;
445
+ }}
446
+ }}
447
+ </style>
448
+ </head>
449
+ <body>
450
+ <div class="container">
451
+ <header>
452
+ <h1>🏠 Property Inventory & Damage Report / تقرير جرد وتحديد تلفيات الممتلكات</h1>
453
+ <p class="sub">AI Matching & Inspection System · نظام المطابقة والفحص الذكي بالذكاء الاصطناعي (DINOv2 + Grounded-SAM + MLLM)</p>
454
+ </header>
455
+
456
+ <div class="summary">
457
+ <div class="stat">
458
+ <div class="num" style="color:#22c55e">{n_intact}</div>
459
+ <div class="lbl">✅ Intact / سليم</div>
460
+ </div>
461
+ <div class="stat">
462
+ <div class="num" style="color:#f59e0b">{n_damaged}</div>
463
+ <div class="lbl">⚠️ Damaged / تالف</div>
464
+ </div>
465
+ <div class="stat">
466
+ <div class="num" style="color:#ef4444">{n_missing}</div>
467
+ <div class="lbl">❌ Missing / مفقود</div>
468
+ </div>
469
+ <div class="stat">
470
+ <div class="num" style="color:#3b82f6">{n_total}</div>
471
+ <div class="lbl">Total Items / إجمالي العناصر</div>
472
+ </div>
473
+ </div>
474
+
475
+ <div class="filters">
476
+ <button class="fbtn active" onclick="filt('ALL', this)">All Items / كل العناصر</button>
477
+ <button class="fbtn" onclick="filt('INTACT', this)">✅ Intact / سليم</button>
478
+ <button class="fbtn" onclick="filt('DAMAGED', this)">⚠️ Damaged / تالف</button>
479
+ <button class="fbtn" onclick="filt('MISSING', this)">❌ Missing / مفقود</button>
480
+ </div>
481
+
482
+ <div id="cards-container">
483
+ {cards}
484
+ </div>
485
+ </div>
486
+
487
+ <script>
488
+ function filt(status, btn) {{
489
+ document.querySelectorAll('.fbtn').forEach(b => b.classList.remove('active'));
490
+ btn.classList.add('active');
491
+ document.querySelectorAll('.card').forEach(c => {{
492
+ const cardStatus = c.getAttribute('data-status');
493
+ if (status === 'ALL') {{
494
+ c.style.display = 'block';
495
+ }} else {{
496
+ c.style.display = cardStatus === status ? 'block' : 'none';
497
+ }}
498
+ }});
499
+ }}
500
+ </script>
501
+ </body>
502
+ </html>"""
503
+
504
+ # Escape curly braces in css and scripts properly (using %% for format double escape or format string)
505
+ report_file = out / "inventory_damage_report.html"
506
+ report_file.write_text(html, encoding="utf-8")
507
+ return str(report_file)
sam_masker.py ADDED
@@ -0,0 +1,106 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import torch
3
+ import numpy as np
4
+ import cv2
5
+ from PIL import Image
6
+
7
+ _dino_processor = None
8
+ _dino_model = None
9
+ _sam_processor = None
10
+ _sam_model = None
11
+
12
+ def get_models(device):
13
+ global _dino_processor, _dino_model, _sam_processor, _sam_model
14
+ if _dino_model is None:
15
+ print("Loading Grounding DINO (Detector)...")
16
+ from transformers import AutoProcessor, AutoModelForZeroShotObjectDetection
17
+ _dino_processor = AutoProcessor.from_pretrained("IDEA-Research/grounding-dino-base")
18
+ _dino_model = AutoModelForZeroShotObjectDetection.from_pretrained("IDEA-Research/grounding-dino-base").to(device)
19
+ _dino_model.eval()
20
+
21
+ if _sam_model is None:
22
+ print("Loading SAM (Segment Anything)...")
23
+ from transformers import SamModel, SamProcessor
24
+ _sam_processor = SamProcessor.from_pretrained("facebook/sam-vit-base")
25
+ _sam_model = SamModel.from_pretrained("facebook/sam-vit-base").to(device)
26
+ _sam_model.eval()
27
+
28
+ return _dino_processor, _dino_model, _sam_processor, _sam_model
29
+
30
+ def generate_damage_mask(image_path, target_phrases_list, output_path):
31
+ """
32
+ Segment the specified damage phrases in the image using Grounding DINO and SAM.
33
+ Saves the final visualized image to output_path.
34
+ Returns True on success, False on failure.
35
+ """
36
+ try:
37
+ device = "cuda" if torch.cuda.is_available() else "cpu"
38
+ d_proc, d_model, s_proc, s_model = get_models(device)
39
+
40
+ # Load image
41
+ image = Image.open(image_path).convert("RGB")
42
+ image_np = np.array(image)
43
+
44
+ # Format query for Grounding DINO (dot-separated)
45
+ text_prompt = ". ".join(target_phrases_list) + "."
46
+
47
+ # Run detector
48
+ inputs = d_proc(images=image, text=text_prompt, return_tensors="pt").to(device)
49
+ with torch.no_grad():
50
+ outputs = d_model(**inputs)
51
+
52
+ results = d_proc.post_process_grounded_object_detection(
53
+ outputs,
54
+ inputs.input_ids,
55
+ threshold=0.25,
56
+ text_threshold=0.25,
57
+ target_sizes=[image.size[::-1]]
58
+ )[0]
59
+
60
+ boxes = results["boxes"]
61
+ scores = results["scores"]
62
+ defect_count = len(boxes)
63
+
64
+ if defect_count == 0:
65
+ # If no damage is located, save the original image and return False
66
+ cv2.imwrite(output_path, cv2.cvtColor(image_np, cv2.COLOR_RGB2BGR))
67
+ return False
68
+
69
+ # Run Segment Anything Model
70
+ box_list = [box.tolist() for box in boxes]
71
+ sam_inputs = s_proc(image, input_boxes=[box_list], return_tensors="pt").to(device)
72
+ with torch.no_grad():
73
+ sam_outputs = s_model(**sam_inputs)
74
+
75
+ pred_masks = sam_outputs.pred_masks.squeeze(1)
76
+ masks = s_proc.image_processor.post_process_masks(
77
+ pred_masks,
78
+ sam_inputs["original_sizes"],
79
+ sam_inputs["reshaped_input_sizes"]
80
+ )[0]
81
+
82
+ # Visualize detections and segmentations
83
+ annotated_img = image_np.copy()
84
+ for i in range(defect_count):
85
+ box = boxes[i].cpu().numpy()
86
+ # Draw green bounding box
87
+ cv2.rectangle(annotated_img, (int(box[0]), int(box[1])), (int(box[2]), int(box[3])), (0, 255, 0), 3)
88
+
89
+ # Extract SAM mask
90
+ mask = masks[i][0].cpu().numpy()
91
+ covered_color = np.zeros_like(annotated_img)
92
+ covered_color[mask] = [255, 0, 0] # Red mask overlay
93
+
94
+ alpha = 0.5
95
+ mask_indices = mask > 0
96
+ annotated_img[mask_indices] = annotated_img[mask_indices] * (1 - alpha) + covered_color[mask_indices] * alpha
97
+
98
+ # Score label
99
+ score = scores[i].item()
100
+ cv2.putText(annotated_img, f"{score:.2f}", (int(box[0]), int(box[1]-10)), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 255, 0), 2)
101
+
102
+ cv2.imwrite(output_path, cv2.cvtColor(annotated_img, cv2.COLOR_RGB2BGR))
103
+ return True
104
+ except Exception as e:
105
+ print(f"Error in generate_damage_mask: {e}")
106
+ return False