feat: intelligently determine necessary views for thin objects and update SparkPlug UI to support dynamic number of views
Build and Deploy / build-and-push (push) Successful in 32s

This commit is contained in:
AI Bot
2026-09-02 14:33:16 +05:30
parent 857648292f
commit 0865b02d17
2 changed files with 27 additions and 8 deletions
+22 -2
View File
@@ -383,9 +383,29 @@ app.post('/api/sparkplug/generate-images', authenticate, async (req, res) => {
// Use the Gemini 3.1 Flash Image API // Use the Gemini 3.1 Flash Image API
const baseUrl = 'https://generativelanguage.googleapis.com/v1beta/models/gemini-3.1-flash-image:generateContent'; const baseUrl = 'https://generativelanguage.googleapis.com/v1beta/models/gemini-3.1-flash-image:generateContent';
const views = ['Front view', 'Back view', 'Left side view', 'Right side view']; // Intelligently determine needed views using a text model
const genAI = new GoogleGenerativeAI(apiKey);
const textModel = genAI.getGenerativeModel({ model: "gemini-3.5-flash-lite" });
const viewPrompt = `You are a 3D modeling assistant. The user wants to generate orthographic views for a product based on this description: "${prompt}".
If the product is extremely thin and flat (like a sachet, pouch, packet, paper, or card), we only need the front and back views.
Otherwise, for a standard 3D volume (bottle, box, can, etc), we need front, back, left, and right.
Return ONLY a valid JSON array of strings representing the views needed.
Example 1: ["Front view", "Back view"]
Example 2: ["Front view", "Back view", "Left side view", "Right side view"]`;
// Run parallel generation for all 4 views let views = ['Front view', 'Back view', 'Left side view', 'Right side view'];
try {
const viewResult = await textModel.generateContent(viewPrompt);
const textResponse = viewResult.response.text().trim();
const match = textResponse.match(/\[.*\]/s);
if (match) {
views = JSON.parse(match[0]);
}
} catch (e) {
console.error("Failed to intelligently determine views, falling back to 4 views", e);
}
// Run parallel generation for determined views
const promises = views.map(async (view, index) => { const promises = views.map(async (view, index) => {
const fullPrompt = `A clean, isolated 3D render concept art of: ${prompt}. const fullPrompt = `A clean, isolated 3D render concept art of: ${prompt}.
Focus ONLY on the object itself. Focus ONLY on the object itself.
+4 -5
View File
@@ -80,12 +80,11 @@ export default function SparkPlug() {
} }
// Call our backend endpoint which queries Gemini Imagen // Call our backend endpoint which queries Gemini Imagen
const images = await api.generateImages(extractedPrompt); const images = await api.generateImages(extractedPrompt);
if (images && images.length > 0) {
if (images && images.length === 4) {
setGeneratedImages(images); setGeneratedImages(images);
setCurrentStep(3); setCurrentStep(3);
} else { } else {
throw new Error("Failed to generate exactly 4 images."); throw new Error("Failed to generate images.");
} }
} catch (err: any) { } catch (err: any) {
console.error(err); console.error(err);
@@ -328,8 +327,8 @@ export default function SparkPlug() {
{generatedImages.map((img, idx) => ( {generatedImages.map((img, idx) => (
<div key={idx} className="relative bg-zinc-900 rounded-lg overflow-hidden border border-zinc-800 flex items-center justify-center"> <div key={idx} className="relative bg-zinc-900 rounded-lg overflow-hidden border border-zinc-800 flex items-center justify-center">
<img src={img} alt={`View ${idx + 1}`} className="max-w-full max-h-[150px] object-contain" /> <img src={img} alt={`View ${idx + 1}`} className="max-w-full max-h-[150px] object-contain" />
<div className="absolute top-2 left-2 bg-black/50 backdrop-blur-sm px-2 py-1 rounded text-[10px] font-bold text-zinc-300"> <div className="absolute top-2 left-2 bg-black/50 backdrop-blur-sm px-2 py-1 rounded text-[10px] font-bold text-zinc-300 uppercase">
{['FRONT', 'BACK', 'LEFT', 'RIGHT'][idx]} {idx === 0 ? 'FRONT' : idx === 1 ? 'BACK' : idx === 2 ? 'LEFT' : 'RIGHT'}
</div> </div>
</div> </div>
))} ))}