#!/usr/bin/env python3 """ Large Action Model - Interface pour self-operating-computer Intégration propre InternVL3 + Nemotron (thinking mode) """ import torch from transformers import AutoProcessor, AutoModelForImageTextToText from PIL import Image import json import time from typing import Dict, List, Any, Optional, Union import re class LargeActionModel: """Large Action Model pour self-operating-computer""" def __init__(self, model_path: str = None): # Charger uniquement le modèle LAM fusionné self.model_path = model_path or "/home/jerem/self-operating-computer/operate/models/merging/unified_model/lam_model.pt" self.processor = None self.model = None self.thinking_mode = True self.loaded = False def load(self) -> bool: """Charger le LAM fusionné""" try: print("🔄 Chargement Large Action Model fusionné...") print(f"📁 Chemin: {self.model_path}") # Charger le modèle LAM fusionné (.pt file) self.model = torch.load(self.model_path, map_location='cpu') print(f"✅ LAM fusionné chargé: {len(self.model)} composants") # Pour le processor, utiliser InternVL3 original en HuggingFace processor_path = "OpenGVLab/InternVL3-8B" print(f"🔄 Chargement processor depuis {processor_path}...") self.processor = AutoProcessor.from_pretrained( processor_path, trust_remote_code=True ) self.loaded = True print("✅ LAM complet chargé et prêt") return True except Exception as e: print(f"❌ Erreur chargement LAM: {e}") return False def set_thinking_mode(self, enabled: bool = True): """Activer/désactiver le mode thinking""" self.thinking_mode = enabled def analyze_screen(self, image: Union[str, Image.Image], question: str = None) -> Dict[str, Any]: """Analyser une capture d'écran""" if not self.loaded: return {"error": "LAM non chargé"} # Préparer l'image if isinstance(image, str): image = Image.open(image) # Question par défaut if not question: question = "Analyze this screen. What UI elements do you see? What actions can I perform?" # Ajouter thinking prompt si activé if self.thinking_mode: question = f"Think step by step. {question} Consider the visual elements, their purpose, and possible interactions." return self._process_vision_request(image, question) def generate_action(self, image: Union[str, Image.Image], task: str) -> Dict[str, Any]: """Générer une action spécifique pour une tâche""" if not self.loaded: return {"error": "LAM non chargé"} if isinstance(image, str): image = Image.open(image) # Prompt action-oriented action_prompt = f""" Task: {task} Analyze this screen and determine the exact action needed to complete this task. Provide: 1. The type of action (click, type, scroll, etc.) 2. The target element description 3. Specific coordinates or text if applicable 4. Step-by-step reasoning if thinking mode is enabled Be precise and actionable. """ result = self._process_vision_request(image, action_prompt) # Post-traiter pour extraire action structurée if "response" in result: action = self._extract_action_from_response(result["response"]) result["structured_action"] = action return result def _process_vision_request(self, image: Image.Image, question: str) -> Dict[str, Any]: """Traiter une requête vision+texte""" start_time = time.time() try: messages = [{ "role": "user", "content": [ {"type": "image", "image": image}, {"type": "text", "text": question} ] }] # Préparer input text = self.processor.apply_chat_template( messages, tokenize=False, add_generation_prompt=True ) inputs = self.processor( text=text, images=image, return_tensors="pt" ) inputs = inputs.to(self.model.device) if 'pixel_values' in inputs: inputs['pixel_values'] = inputs['pixel_values'].to(torch.bfloat16) # Génération with torch.no_grad(): output = self.model.generate( **inputs, max_new_tokens=300, do_sample=False, temperature=0.1 ) response = self.processor.decode(output[0], skip_special_tokens=True) # Extraire réponse LAM if "assistant" in response: lam_response = response.split("assistant")[-1].strip() else: lam_response = response.split("user")[-1].strip() inference_time = time.time() - start_time return { "response": lam_response, "inference_time": inference_time, "thinking_mode": self.thinking_mode, "success": True } except Exception as e: return { "error": str(e), "success": False, "inference_time": time.time() - start_time } def _extract_action_from_response(self, response: str) -> Dict[str, Any]: """Extraire action structurée de la réponse texte""" action = { "type": "analyze", # Par défaut "confidence": 0.5, "description": response, "coordinates": None, "text_input": None } response_lower = response.lower() # Détecter type d'action if "click" in response_lower: action["type"] = "click" action["confidence"] = 0.8 # Chercher coordonnées si mentionnées coord_match = re.search(r'(\d+),\s*(\d+)', response) if coord_match: action["coordinates"] = [int(coord_match.group(1)), int(coord_match.group(2))] elif any(word in response_lower for word in ["type", "enter", "input", "write"]): action["type"] = "type" action["confidence"] = 0.8 # Chercher texte à saisir type_match = re.search(r'"([^"]*)"', response) if type_match: action["text_input"] = type_match.group(1) elif "scroll" in response_lower: action["type"] = "scroll" action["confidence"] = 0.7 direction = "down" # Par défaut if "up" in response_lower: direction = "up" action["direction"] = direction # Augmenter confiance si éléments spécifiques détectés if any(word in response_lower for word in ["button", "link", "menu", "field"]): action["confidence"] = min(action["confidence"] + 0.1, 0.9) return action def get_status(self) -> Dict[str, Any]: """Status du LAM""" return { "loaded": self.loaded, "thinking_mode": self.thinking_mode, "model_path": self.model_path, "device": str(self.model.device) if self.model else None } # Interface compatibilité avec le code existant def get_next_action(screenshot_path: str, objective: str, model_name: str = "lam") -> str: """Interface pour compatibilité avec operate.py existant""" # Instance globale LAM if not hasattr(get_next_action, 'lam'): get_next_action.lam = LargeActionModel() if not get_next_action.lam.load(): return "Error: Could not load LAM" try: # Analyser avec LAM result = get_next_action.lam.generate_action(screenshot_path, objective) if not result.get("success", False): return f"Error: {result.get('error', 'Unknown error')}" # Formater réponse compatible response = result["response"] # Ajouter action structurée si disponible if "structured_action" in result: action = result["structured_action"] response += f"\n\nAction: {action['type']}" if action['coordinates']: response += f" at {action['coordinates']}" if action['text_input']: response += f" with text: '{action['text_input']}'" return response except Exception as e: return f"LAM Error: {str(e)}" def main(): """Test du LAM""" print("🤖 Large Action Model - Test Interface") lam = LargeActionModel() if not lam.load(): return # Test avec screenshot existant screenshot_path = "/home/jerem/self-operating-computer/Screenshot from 2025-08-14 09-50-26.png" try: # Test 1: Analyse générale print("\n🔍 Test 1: Analyse générale") result = lam.analyze_screen(screenshot_path) print(f"Réponse: {result['response'][:200]}...") print(f"Temps: {result['inference_time']:.2f}s") # Test 2: Action spécifique print("\n🎯 Test 2: Action spécifique") result = lam.generate_action(screenshot_path, "Click on the contact button") print(f"Action: {result['structured_action']}") print(f"Réponse: {result['response'][:200]}...") except Exception as e: print(f"Erreur test: {e}") if __name__ == "__main__": main()