Token size for contact sheet

This commit is contained in:
moosecrap 2026-07-22 22:02:52 -07:00
parent 5af30f8f2d
commit d87fcb85be
2 changed files with 62 additions and 60 deletions

View File

@ -18,14 +18,14 @@ class Tool:
# Import handlers from separate files
from .read_image import handle as read_image_handler
from .read_metadata import handle as read_png_metadata_handler
from .list_thumbnails import handle as list_thumbnails_handler
from .contact_sheet import handle as contact_sheet_handler
from .list_directory import handle as list_directory_details_handler
# Central registry of all available tools
TOOL_REGISTRY = [
Tool(
name="read_image",
description="Reads an image from the disk and returns it as an image content object.",
description="Reads an image from the disk and returns it as a full size image content object.",
schema={
"type": "object",
"properties": {"path": {"type": "string", "description": "Path to the image file"}},
@ -44,23 +44,22 @@ TOOL_REGISTRY = [
handler=read_png_metadata_handler
),
Tool(
name="list_thumbnails",
description="Generates a thumbnail grid of images in a directory. Supports pagination and sorting.",
name="contact_sheet",
description="Generates a high-density contact sheet of images in a directory, for low-token previews.",
schema={
"type": "object",
"properties": {
"path": {"type": "string", "description": "Path to the directory containing images"},
"page": {"type": "integer", "description": "The page number to retrieve (1-indexed)", "default": 1},
"page_size": {"type": "integer", "description": "Number of images per page", "default": 64},
"sort_by": {"type": "string", "description": "Sort order: 'mtime' (newest first, default), 'name' (alphabetical), or 'size' (largest first)", "default": "mtime"},
},
"required": ["path"],
},
handler=list_thumbnails_handler
handler=contact_sheet_handler
),
Tool(
name="list_directory",
description="Lists the contents of a directory in a file browser format. Directories first, then files. Supports pagination and sorting.",
description="Lists the contents of a directory in a simplified format. Directories first, then files. Supports pagination and sorting.",
schema={
"type": "object",
"properties": {

View File

@ -1,23 +1,25 @@
import os
import base64
import mimetypes
import logging
import asyncio
import io
import datetime
import time
from typing import Any, Dict
from typing import Any, Callable, Dict, List, Union
from PIL import Image, ImageDraw, ImageFont
from tools.utils import format_relative_time
logger = logging.getLogger("MattCP")
async def handle(args: Dict[str, Any]):
"""
Generates a thumbnail grid of images in a directory.
Returns a combined text list (with global indices) and a visual contact sheet.
Generates a contact sheet of images.
Grid: 10 columns x 7 rows (70 images total).
Sized for optimal token usage (1120 tokens) on llama.cpp.
"""
dir_path = args.get("path")
page = int(args.get("page", 1))
page_size = int(args.get("page_size", 64))
sort_by = args.get("sort_by", "mtime")
if not dir_path:
@ -27,8 +29,6 @@ async def handle(args: Dict[str, Any]):
exts = ('.png', '.jpg', '.jpeg', '.webp', '.bmp')
file_info_list = []
# 1. Gather all valid images and their stats
for f in os.listdir(dir_path):
if f.lower().endswith(exts):
full_path = os.path.join(dir_path, f)
@ -46,41 +46,39 @@ async def handle(args: Dict[str, Any]):
if not file_info_list:
return {"text": "No supported images found in the directory."}
# 2. Sort the list based on requested criteria
# Sorting
if sort_by == "mtime":
file_info_list.sort(key=lambda x: x["mtime"], reverse=True)
sort_label = "Date (newest first)"
elif sort_by == "size":
file_info_list.sort(key=lambda x: x["size"], reverse=True)
sort_label = "Size (largest first)"
else:
file_info_list.sort(key=lambda x: x["name"].lower())
sort_label = "Name (alphabetical)"
# 3. Apply Pagination
# Fixed grid dimensions for token optimization
COLS = 10
ROWS = 7
PAGE_SIZE = COLS * ROWS
total_files = len(file_info_list)
start_idx = (page - 1) * page_size
end_idx = start_idx + page_size
start_idx = (page - 1) * PAGE_SIZE
end_idx = start_idx + PAGE_SIZE
paged_files = file_info_list[start_idx:end_idx]
if not paged_files:
return {"text": f"No images found on page {page}."}
# Layout constants
thumb_size = 160
padding = 15
label_height = 35
thumb_size = 192 # 192 / 48 = 4 patches per side
num_files = len(paged_files)
cols = int(num_files**0.5) if num_files > 0 else 1
if cols == 0: cols = 1
rows = (num_files + cols - 1) // cols
canvas_w = COLS * thumb_size
canvas_h = ROWS * thumb_size
canvas_w = cols * (thumb_size + padding) + padding
canvas_h = rows * (thumb_size + label_height + padding) + padding
# Create dark gray background
canvas = Image.new('RGB', (canvas_w, canvas_h), (30, 30, 30))
draw = ImageDraw.Draw(canvas)
# Attempt to load a bold system font for the indices
# Font loading
font = None
font_paths = [
"/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf",
@ -90,70 +88,75 @@ async def handle(args: Dict[str, Any]):
for path in font_paths:
if os.path.exists(path):
try:
font = ImageFont.truetype(path, 20)
font = ImageFont.truetype(path, 24)
break
except Exception:
continue
if font is None:
font = ImageFont.load_default()
file_details = []
# 4. Paste thumbnails and render indices
for i, info in enumerate(paged_files):
filename = info["name"]
full_path = info["path"]
row = i // cols
col = i % cols
row = i // COLS
col = i % COLS
x = padding + col * (thumb_size + padding)
y = padding + row * (thumb_size + label_height + padding)
x = col * thumb_size
y = row * thumb_size
# Continuous indexing across pages
global_index = start_idx + i + 1
try:
with Image.open(full_path) as img:
width, height = img.size
img.thumbnail((thumb_size, thumb_size))
# Center image in its slot
off_x = (thumb_size - img.width) // 2
off_y = (thumb_size - img.height) // 2
canvas.paste(img, (x + off_x, y + off_y))
# Format text metadata
from tools.utils import format_relative_time # Import helper from utils
mod_time_rel = format_relative_time(info["mtime"])
size_kb = info["size"] / 1024
file_details.append(f"{global_index}. {filename} | {width}x{height} | {size_kb:.1f}KB | {mod_time_rel}")
# Draw the index number centered under the image
# Index number in lower-left corner
text = str(global_index)
if font:
bbox = draw.textbbox((0, 0), text, font=font)
text_w = bbox[2] - bbox[0]
text_h = bbox[3] - bbox[1]
else:
text_w = len(text) * 7
text_w = len(text) * 8
text_h = 15
text_x = x + (thumb_size - text_w) // 2
draw.text((text_x, y + thumb_size + 2), text, fill=(255, 255, 255), font=font)
# Semi-opaque background rectangle for the number
draw.rectangle([x, y + thumb_size - text_h - 5, x + text_w + 5, y + thumb_size - 2], fill=(0, 0, 0, 180))
draw.text((x + 2, y + thumb_size - text_h - 7), text, fill=(255, 255, 255), font=font)
except Exception as e:
logger.error(f"Failed to process {filename}: {e}")
draw.text((x, y + thumb_size // 2), "Error", fill=(255, 0, 0), font=font)
file_details.append(f"{global_index}. {filename} | ERROR")
logger.error(f"Failed to process {info['name']}: {e}")
draw.text((x + 5, y + thumb_size // 2), "Error", fill=(255, 0, 0), font=font)
# 5. Encode result to base64 PNG
buf = io.BytesIO()
canvas.save(buf, format='PNG')
img_data = base64.b64encode(buf.getvalue()).decode("utf-8")
total_pages = (total_files + page_size - 1) // page_size
header_text = f"Directory: {dir_path}\nPage {page} of {total_pages} ({total_files} total images). Showing {start_idx+1}-{min(end_idx, total_files)}."
total_pages = (total_files + PAGE_SIZE - 1) // PAGE_SIZE
details_text = "\n".join(file_details)
# Range shown: first and last of the CURRENT PAGE
first_on_page = paged_files[0]
last_on_page = paged_files[-1]
if sort_by == "mtime":
first_val = f"{format_relative_time(first_on_page['mtime'])} ({datetime.datetime.fromtimestamp(first_on_page['mtime']).strftime('%Y-%m-%d')})"
last_val = f"{format_relative_time(last_on_page['mtime'])} ({datetime.datetime.fromtimestamp(last_on_page['mtime']).strftime('%Y-%m-%d')})"
elif sort_by == "size":
first_val = f"{first_on_page['size']/1024:.1f}KB"
last_val = f"{last_on_page['size']/1024:.1f}KB"
else:
first_val = first_on_page['name']
last_val = last_on_page['name']
header_text = (
f"Directory: {dir_path}\n"
f"Page {page} of {total_pages} ({total_files} images total). Sorted by: {sort_label}\n"
f"Range shown: {first_val} ... {last_val}"
)
return [
{"type": "text", "text": f"{header_text}\n\nFile List:\n{details_text}"},
{"type": "text", "text": header_text},
{"type": "image", "data": img_data, "mimeType": "image/png"}
]