Train notebook
No files matched your search
@@ -1,5 +1,5 @@
|
|||||||
"""
|
"""
|
||||||
STEP 1 ─ prepare_dataset.py
|
STEP 1 ─ 1_prepare_dataset.py
|
||||||
===========================
|
===========================
|
||||||
Run this FIRST. It creates the required folder structure and
|
Run this FIRST. It creates the required folder structure and
|
||||||
a dataset.yaml that YOLOv8 expects.
|
a dataset.yaml that YOLOv8 expects.
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
"""
|
"""
|
||||||
STEP 2 — augment.py
|
STEP 2 — 2_augment.py
|
||||||
====================
|
====================
|
||||||
Expands your small labeled dataset using augmentations tuned
|
Expands your small labeled dataset using augmentations tuned
|
||||||
specifically for retail shelf / packaged-goods detection.
|
specifically for retail shelf / packaged-goods detection.
|
||||||
@@ -0,0 +1,224 @@
|
|||||||
|
"""
|
||||||
|
SETP 4 - detect.py
|
||||||
|
====================
|
||||||
|
Run detection on a single image, a folder, or a webcam stream.
|
||||||
|
Draw bounding boxes, prints a per-class count summary, and
|
||||||
|
saves an annotated result image.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python detect.py --source scene.jpg
|
||||||
|
python detect.py --source images/
|
||||||
|
python detect.py --source 0
|
||||||
|
python detect.py --source scene.jpg --conf 0.5 --wieghts best.pt
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
import cv2
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
from collections import defaultdict
|
||||||
|
from ultralytics import YOLO
|
||||||
|
|
||||||
|
# ── Defaults ──────────────────────────────────────────────────────────────────
|
||||||
|
DEFAULT_WEIGHTS = "runs/detect/coffee_v1/weights/best.pt"
|
||||||
|
DEFAULT_CONF = .35 # lower = more detections and more false positives.
|
||||||
|
DEFAULT_IOU = .45 # NMS IoU threshold.
|
||||||
|
DEFAULT_IMGSZ = 640
|
||||||
|
OUTPUT_DIR = Path("detections")
|
||||||
|
|
||||||
|
|
||||||
|
# ── Colour palette (one BGR colour per class index) ──────────────────────────
|
||||||
|
PALETTE = [
|
||||||
|
( 0, 200, 0), # green
|
||||||
|
( 0, 120, 255), # orange
|
||||||
|
(255, 50, 50), # blue
|
||||||
|
(200, 0, 200), # magenta
|
||||||
|
( 0, 200, 200), # yellow
|
||||||
|
( 80, 200, 80),
|
||||||
|
(200, 100, 0),
|
||||||
|
( 0, 80, 200),
|
||||||
|
(150, 0, 150),
|
||||||
|
(100, 200, 0),
|
||||||
|
( 0, 150, 150),
|
||||||
|
(200, 50, 100),
|
||||||
|
( 50, 50, 200),
|
||||||
|
]
|
||||||
|
|
||||||
|
def get_color(class_id: int):
|
||||||
|
return PALETTE[class_id % len(PALETTE)]
|
||||||
|
|
||||||
|
# ── Drawing ──────────────────────────
|
||||||
|
def draw_box(img, box, class_name, conf, class_id):
|
||||||
|
x1, y1, x2, y2 = map(int, box)
|
||||||
|
color = get_color(class_id)
|
||||||
|
|
||||||
|
cv2.rectangle(img, (x1, y1), (x2, y2), color, 2)
|
||||||
|
|
||||||
|
label = f"{class_name} {conf: .0%}"
|
||||||
|
(tw, th), baseline = cv2.getTextSize(label, cv2.FONT_HERSHEY_SIMPLEX, 0.55, 1)
|
||||||
|
cv2.rectangle(
|
||||||
|
img,
|
||||||
|
(x1, y1 - th - baseline - 6),
|
||||||
|
(x1 + tw + 4, y1),
|
||||||
|
color, cv2.FILLED
|
||||||
|
)
|
||||||
|
|
||||||
|
cv2.putText(
|
||||||
|
img,
|
||||||
|
label,
|
||||||
|
(x1 + 2, y1 - baseline -2),
|
||||||
|
cv2.FONT_HERSHEY_SIMPLEX,
|
||||||
|
0.55,
|
||||||
|
(0, 0, 0), 1, cv2.LINE_AA
|
||||||
|
)
|
||||||
|
|
||||||
|
def draw_summary(img, counts: dict):
|
||||||
|
""" Overlay a product count table in the top-right corner."""
|
||||||
|
if not counts:
|
||||||
|
return
|
||||||
|
|
||||||
|
lines = ["── Count ──"] + [f"{n:>2}x {name}" for name, n in sorted(counts.items())]
|
||||||
|
|
||||||
|
x = img.shape[1] - 200
|
||||||
|
y = 16
|
||||||
|
for line in lines:
|
||||||
|
(tw, th), _ = cv2.getTextSize(line, cv2.FONT_HERSHEY_SIMPLEX, 0.5, 1)
|
||||||
|
cv2.rectangle(img, (x - 4, y - th - 2), (x + tw + 4, y + 4), (30, 30, 30), cv2.FILLED)
|
||||||
|
cv2.putText(img, line, (x, y), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (240, 240, 240), 1, cv2.LINE_AA)
|
||||||
|
y += th + 8
|
||||||
|
|
||||||
|
|
||||||
|
# ── Single image inference ────────────────────────────────────────────────────
|
||||||
|
def detect_image(model, img_path: Path, conf: float, iou: float, imgsz: int):
|
||||||
|
img = cv2.imread(str(img_path))
|
||||||
|
if img is None:
|
||||||
|
print(f"[WARN] Cannot read {img_path}")
|
||||||
|
return
|
||||||
|
results = model.predict(
|
||||||
|
source=str(img_path),
|
||||||
|
conf=conf,
|
||||||
|
iou=iou,
|
||||||
|
imgsz=imgsz,
|
||||||
|
verbose=False,
|
||||||
|
)[0]
|
||||||
|
|
||||||
|
counts = defaultdict(int)
|
||||||
|
|
||||||
|
for box in results.boxes:
|
||||||
|
cid = int(box.cls)
|
||||||
|
name = model.names[cid]
|
||||||
|
score = float(box.conf)
|
||||||
|
draw_box(img, box.xyxy[0].tolist(), name, score, cid)
|
||||||
|
counts[name] += 1
|
||||||
|
|
||||||
|
draw_summary(img, counts)
|
||||||
|
|
||||||
|
OUTPUT_DIR.mkdir(exist_ok=True)
|
||||||
|
out_path = OUTPUT_DIR / img_path.name
|
||||||
|
cv2.imwrite(str(out_path), img)
|
||||||
|
|
||||||
|
# Console Summary
|
||||||
|
print(f"\n{img_path.name}")
|
||||||
|
if counts:
|
||||||
|
for name, n in sorted(counts.items()):
|
||||||
|
print(f" {n:>2}x {name}")
|
||||||
|
else:
|
||||||
|
print(" (no detections)")
|
||||||
|
print(f" -> saved: {out_path}")
|
||||||
|
|
||||||
|
return img, counts
|
||||||
|
|
||||||
|
# ── Webcam / video stream ─────────────────────────────────────────────────────
|
||||||
|
def detect_stream(model, source, conf: float, iou: float, imgsz: int):
|
||||||
|
cap = cv2.VideoCapture(int(source) if source.isdigit() else source)
|
||||||
|
if not cap.isOpened():
|
||||||
|
print(f"[ERROR] Cannot open source: {source}")
|
||||||
|
return
|
||||||
|
|
||||||
|
print("Streaming - press Q to quit")
|
||||||
|
while True:
|
||||||
|
ret, frame = cap.read()
|
||||||
|
if not ret:
|
||||||
|
break
|
||||||
|
|
||||||
|
results = model.predict(
|
||||||
|
source = frame,
|
||||||
|
conf = conf,
|
||||||
|
iou = iou,
|
||||||
|
imgsz = imgsz,
|
||||||
|
verbose = False
|
||||||
|
)[0]
|
||||||
|
|
||||||
|
counts = defaultdict(int)
|
||||||
|
for box in results.boxes:
|
||||||
|
cid = int(box.cls)
|
||||||
|
name = model.names[cid]
|
||||||
|
draw_box(frame, box.xyxy[0].tolist(), name, float(box.conf), cid)
|
||||||
|
counts[name] += 1
|
||||||
|
|
||||||
|
draw_summary(frame, counts)
|
||||||
|
cv2.imshow("Coffee Detector", frame)
|
||||||
|
if cv2.waitKey(1) & 0xFF == ord("q"):
|
||||||
|
break
|
||||||
|
|
||||||
|
cap.release()
|
||||||
|
cv2.destroyAllWindows()
|
||||||
|
|
||||||
|
|
||||||
|
# ── Entry point ─────────────────────────────────────────────────────
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Coffee Product Detector")
|
||||||
|
parser.add_argument("--source" , default="scene.jpg" , help="Image path, folder, or 0 for webcam")
|
||||||
|
parser.add_argument("--weights", default=DEFAULT_WEIGHTS, help="Path to *.pt")
|
||||||
|
parser.add_argument("--conf" , default=DEFAULT_CONF , type=float, help="Confidence threshold (0-1)")
|
||||||
|
parser.add_argument("--iou" , default=DEFAULT_IOU , type=float, help="NMS IoU threshold (0-1)")
|
||||||
|
parser.add_argument("--imgsz" , default=DEFAULT_IMGSZ , type=int , help="Inference image size")
|
||||||
|
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
if not Path(args.weights).exists():
|
||||||
|
print(f"[ERROR] Weights not found: {args.weights}")
|
||||||
|
print(" Run train.py first.")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
print(f"Loading model: {args.weights}")
|
||||||
|
model = YOLO(args.weights)
|
||||||
|
print(f"Classes: {list(model.names.values())}\n")
|
||||||
|
|
||||||
|
source = Path(args.source)
|
||||||
|
|
||||||
|
# Webcam / video
|
||||||
|
if args.source.isdigit() or str(source).endswith((".mp4", ".avi", ".mov")):
|
||||||
|
detect_stream(model, args.source, args.conf, args.iou, args.imgsz)
|
||||||
|
return
|
||||||
|
|
||||||
|
# Folder
|
||||||
|
if source.is_dir():
|
||||||
|
exts = {".jpg", ".jpeg", ".png", ".bmp", ".webp"}
|
||||||
|
images = sorted([p for p in source.iterdir() if p.suffix.lower() in exts])
|
||||||
|
print(f"Processing {len(images)} images from {source}...")
|
||||||
|
total_counts = defaultdict(int)
|
||||||
|
for img_path in images:
|
||||||
|
_, counts = detect_image(model, img_path, args.conf, args.iou, args.imgsz)
|
||||||
|
if counts:
|
||||||
|
for k, v in counts.items():
|
||||||
|
total_counts[k] += v
|
||||||
|
|
||||||
|
print("\n── Total across all images ──")
|
||||||
|
for name, n in sorted(total_counts.items()):
|
||||||
|
print(f" {n:>3}x {name}")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Single image
|
||||||
|
if source.exists():
|
||||||
|
detect_image(model, source, args.conf, args.iou, args.imgsz)
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"[ERROR] Source not found: {source}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
|
After Width: | Height: | Size: 464 KiB |
|
After Width: | Height: | Size: 414 KiB |
|
After Width: | Height: | Size: 423 KiB |
|
After Width: | Height: | Size: 222 KiB |
|
After Width: | Height: | Size: 326 KiB |
|
After Width: | Height: | Size: 410 KiB |
|
After Width: | Height: | Size: 317 KiB |
|
After Width: | Height: | Size: 410 KiB |
|
After Width: | Height: | Size: 229 KiB |
|
After Width: | Height: | Size: 224 KiB |
|
After Width: | Height: | Size: 271 KiB |
|
After Width: | Height: | Size: 252 KiB |
|
After Width: | Height: | Size: 174 KiB |
|
After Width: | Height: | Size: 229 KiB |
|
After Width: | Height: | Size: 148 KiB |
|
After Width: | Height: | Size: 225 KiB |
|
After Width: | Height: | Size: 317 KiB |
|
After Width: | Height: | Size: 222 KiB |
|
After Width: | Height: | Size: 189 KiB |
|
After Width: | Height: | Size: 414 KiB |
|
After Width: | Height: | Size: 464 KiB |
|
After Width: | Height: | Size: 410 KiB |
|
After Width: | Height: | Size: 326 KiB |
|
After Width: | Height: | Size: 410 KiB |
|
After Width: | Height: | Size: 423 KiB |
|
After Width: | Height: | Size: 212 KiB |
|
After Width: | Height: | Size: 260 KiB |
|
After Width: | Height: | Size: 352 KiB |
|
After Width: | Height: | Size: 287 KiB |
|
After Width: | Height: | Size: 290 KiB |
|
After Width: | Height: | Size: 283 KiB |
|
After Width: | Height: | Size: 188 KiB |
|
After Width: | Height: | Size: 317 KiB |
|
After Width: | Height: | Size: 354 KiB |
|
After Width: | Height: | Size: 374 KiB |
|
After Width: | Height: | Size: 175 KiB |
|
After Width: | Height: | Size: 310 KiB |
|
After Width: | Height: | Size: 189 KiB |
|
After Width: | Height: | Size: 245 KiB |
|
After Width: | Height: | Size: 499 KiB |
|
After Width: | Height: | Size: 308 KiB |
|
After Width: | Height: | Size: 298 KiB |
|
After Width: | Height: | Size: 381 KiB |
|
After Width: | Height: | Size: 332 KiB |
|
After Width: | Height: | Size: 296 KiB |
|
After Width: | Height: | Size: 323 KiB |
|
After Width: | Height: | Size: 263 KiB |
|
After Width: | Height: | Size: 350 KiB |
|
After Width: | Height: | Size: 246 KiB |
|
After Width: | Height: | Size: 326 KiB |
|
After Width: | Height: | Size: 364 KiB |
|
After Width: | Height: | Size: 367 KiB |
|
After Width: | Height: | Size: 243 KiB |
|
After Width: | Height: | Size: 455 KiB |
|
After Width: | Height: | Size: 321 KiB |
|
After Width: | Height: | Size: 409 KiB |
|
After Width: | Height: | Size: 303 KiB |
|
After Width: | Height: | Size: 368 KiB |
|
After Width: | Height: | Size: 369 KiB |
|
After Width: | Height: | Size: 532 KiB |
|
After Width: | Height: | Size: 294 KiB |
|
After Width: | Height: | Size: 413 KiB |
|
After Width: | Height: | Size: 384 KiB |
|
After Width: | Height: | Size: 323 KiB |
@@ -0,0 +1 @@
|
|||||||
|
{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"datasetVersion","sourceId":3771150,"datasetId":2004518,"databundleVersionId":3825728}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ultralytics","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-03T10:28:09.186553Z","iopub.execute_input":"2026-05-03T10:28:09.186784Z","iopub.status.idle":"2026-05-03T10:28:15.175437Z","shell.execute_reply.started":"2026-05-03T10:28:09.186760Z","shell.execute_reply":"2026-05-03T10:28:15.174587Z"}},"outputs":[{"name":"stdout","text":"\u001b[2K \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m1.2/1.2 MB\u001b[0m \u001b[31m20.6 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0ma \u001b[36m0:00:01\u001b[0m\n\u001b[?25h","output_type":"stream"}],"execution_count":1},{"cell_type":"markdown","source":"# Dataset - SKU110K *(most popular)*\n- 11,762 shelf images with ~1.2million annotated products.\n- Dense shelf scenarios\n- [on Github](http://github.com/eg4000/SKU110K_CVPR19) | [on Kaggle](https://www.kaggle.com/datasets/thedatasith/sku110k-annotations)\n\nOther datasets to check:\n- Grocery Store Dataset (Grozi-120)\n- WebMarket\n- RPC (Retail Product Checkout)","metadata":{}},{"cell_type":"code","source":"import os\n\nBASE_PATH = \"/kaggle/input/datasets/thedatasith/sku110k-annotations\"\nDATASET_FOLDER = None\n\n# find SKU110K_fixed folder\nfor item in os.listdir(BASE_PATH):\n if \"SKU110K\" in item:\n DATASET_FOLDER = os.path.join(BASE_PATH, item)\n break\n\nprint(\"📁 Dataset folder:\", DATASET_FOLDER)\n\nfor root, dirs, files in os.walk(DATASET_FOLDER):\n level = root.replace(DATASET_FOLDER, '').count(os.sep)\n indent = ' ' * 2 * level\n print(f\"{indent}📁 {os.path.basename(root)}/\")\n for f in files[:5]:\n print(f\"{indent} 📄 {f}\")\n if level >= 2:\n break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-03T10:28:15.180147Z","iopub.execute_input":"2026-05-03T10:28:15.180435Z","iopub.status.idle":"2026-05-03T10:28:18.766862Z","shell.execute_reply.started":"2026-05-03T10:28:15.180387Z","shell.execute_reply":"2026-05-03T10:28:18.766011Z"}},"outputs":[{"name":"stdout","text":"📁 Dataset folder: /kaggle/input/datasets/thedatasith/sku110k-annotations/SKU110K_fixed\n📁 SKU110K_fixed/\n 📁 labels/\n 📁 val/\n 📄 val_30.txt\n 📄 val_216.txt\n 📄 val_16.txt\n 📄 val_499.txt\n 📄 val_180.txt\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"import yaml\n\nYAML_PATH = BASE_PATH + \"/data_kaggle.yaml\"\n\nwith open(YAML_PATH, \"r\") as f:\n data = yaml.safe_load(f)\n\nprint(\"Classes:\", data.get(\"names\"))\nprint(\"Number of classes:\", len(data.get(\"names\", [])))\nprint(\"Train path:\", data.get(\"train\"))\nprint(\"Val path:\", data.get(\"val\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-03T10:28:18.768355Z","iopub.execute_input":"2026-05-03T10:28:18.768863Z","iopub.status.idle":"2026-05-03T10:28:18.803040Z","shell.execute_reply.started":"2026-05-03T10:28:18.768833Z","shell.execute_reply":"2026-05-03T10:28:18.802326Z"}},"outputs":[{"name":"stdout","text":"Classes: ['object']\nNumber of classes: 1\nTrain path: train\nVal path: val\n","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"import yaml\n\nYAML_PATH = BASE_PATH + \"/data_kaggle.yaml\"\n\nwith open(YAML_PATH, \"r\") as f:\n data = yaml.safe_load(f)\n\n# 🔧 FIX PATHS\ndata[\"train\"] = DATASET_FOLDER + \"/images/train\"\ndata[\"val\"] = DATASET_FOLDER + \"/images/val\"\n\n# save fixed yaml\nFIXED_YAML_PATH = \"/kaggle/working/fixed_data.yaml\"\n\nwith open(FIXED_YAML_PATH, \"w\") as f:\n yaml.dump(data, f)\n\nprint(\"✅ Fixed YAML saved at:\", FIXED_YAML_PATH)\nprint(data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-03T10:28:20.692186Z","iopub.execute_input":"2026-05-03T10:28:20.692762Z","iopub.status.idle":"2026-05-03T10:28:20.701198Z","shell.execute_reply.started":"2026-05-03T10:28:20.692697Z","shell.execute_reply":"2026-05-03T10:28:20.700594Z"}},"outputs":[{"name":"stdout","text":"✅ Fixed YAML saved at: /kaggle/working/fixed_data.yaml\n{'path': '/kaggle/input/sku110k-annotations/SKU110K_fixed/images', 'train': '/kaggle/input/datasets/thedataLine truncated
|
||||||
|
After Width: | Height: | Size: 179 KiB |
|
After Width: | Height: | Size: 161 KiB |
|
After Width: | Height: | Size: 156 KiB |
|
After Width: | Height: | Size: 121 KiB |
|
After Width: | Height: | Size: 171 KiB |
|
After Width: | Height: | Size: 155 KiB |
|
After Width: | Height: | Size: 121 KiB |
|
After Width: | Height: | Size: 165 KiB |
|
After Width: | Height: | Size: 75 KiB |
|
After Width: | Height: | Size: 79 KiB |
|
After Width: | Height: | Size: 86 KiB |
|
After Width: | Height: | Size: 85 KiB |
|
After Width: | Height: | Size: 56 KiB |
|
After Width: | Height: | Size: 83 KiB |
|
After Width: | Height: | Size: 55 KiB |
|
After Width: | Height: | Size: 81 KiB |
|
After Width: | Height: | Size: 121 KiB |
|
After Width: | Height: | Size: 121 KiB |
|
After Width: | Height: | Size: 71 KiB |
|
After Width: | Height: | Size: 161 KiB |
|
After Width: | Height: | Size: 179 KiB |
|
After Width: | Height: | Size: 155 KiB |
|
After Width: | Height: | Size: 171 KiB |
|
After Width: | Height: | Size: 165 KiB |
|
After Width: | Height: | Size: 156 KiB |
|
After Width: | Height: | Size: 87 KiB |
|
After Width: | Height: | Size: 91 KiB |
|
After Width: | Height: | Size: 129 KiB |
|
After Width: | Height: | Size: 105 KiB |