Compare commits
85 Commits
fix/test-s
...
f384fbb749
| Author | SHA1 | Date | |
|---|---|---|---|
| f384fbb749 | |||
| 565bdaa568 | |||
| c641204a6b | |||
| 4b645c6fb2 | |||
| c7c7ce29f8 | |||
| b36dde77d8 | |||
| 93b2140844 | |||
| 67c3d464e0 | |||
| d298f03891 | |||
| 604f2d7341 | |||
| cd8f35056c | |||
| 800fb1da98 | |||
| 6cd068f951 | |||
| f46b0b7bcb | |||
| 5fbbe3d273 | |||
| f85d0a8a76 | |||
| f0a54d4e20 | |||
| 0a73c35809 | |||
| e535c10b65 | |||
| 91effbc843 | |||
| 3da3849ca1 | |||
| 51ee7a6793 | |||
| 936da47f61 | |||
| d1e0995148 | |||
| 2f8eebb7e9 | |||
| 9c6f80de9d | |||
| cff7e976e0 | |||
| f6f15ebd9a | |||
| 1cc367697e | |||
| 738a59ac8d | |||
| cd6cecbe27 | |||
| 4af4ddb060 | |||
| f32ee46d8c | |||
| d2de5f91de | |||
| 9a13216064 | |||
| aa5184786e | |||
| 2e1edec56a | |||
| df18a48a84 | |||
| f1a8573be8 | |||
| da7201117c | |||
| fddf14fd67 | |||
| 392abff313 | |||
| 556bd181fa | |||
| 2c44331f03 | |||
| b83b55e02b | |||
| db226ed7c2 | |||
| 47f94b699c | |||
| 96fdbd7db7 | |||
| ca91ae4b33 | |||
| a560225dc9 | |||
| 849fb63426 | |||
| fc44633ebc | |||
| 0f5b71708d | |||
| 0ef2840f79 | |||
| 068a6a616a | |||
| effb1f5ae1 | |||
| 0e43996ccd | |||
| b6846ab0fe | |||
| 6db579f45b | |||
| 0ed12303ac | |||
| 6abb519e3b | |||
| 4e91db01c9 | |||
| e55abc5a8a | |||
| 03105437b8 | |||
| b9c29a5a2d | |||
| 71310b8e84 | |||
| 073a90c38c | |||
| 44fae37cc7 | |||
| 48071cc9b8 | |||
| 10a85a91f1 | |||
| 5bf0053884 | |||
| a846462d02 | |||
| 0dbafd0a82 | |||
| 83e5b94ddf | |||
| dd8285e1ce | |||
| ac5d5351a6 | |||
| ad012b4cd4 | |||
| 5fcf1f180b | |||
| 9a74d89477 | |||
| dc4b576bc1 | |||
| e94dfe8c5c | |||
| 7aa6bfccf6 | |||
| 5fef014cb4 | |||
| 0bdfd999d2 | |||
| 4ad559e107 |
11
.gitignore
vendored
11
.gitignore
vendored
@@ -13,6 +13,9 @@
|
||||
!tests/fixtures/*.xml
|
||||
!tests/fixtures/*.jpg
|
||||
!tests/fixtures/*.json
|
||||
!tests/e2e/fixtures/*.xml
|
||||
!tests/e2e/fixtures/*.jpg
|
||||
!tests/e2e/fixtures/*.json
|
||||
logs/
|
||||
*.pyc
|
||||
__pycache__/
|
||||
@@ -28,8 +31,14 @@ Pipfile.lock
|
||||
*.ini
|
||||
*.db
|
||||
|
||||
# Debug artifacts
|
||||
# Debug artifacts & garbage scripts (Rule 5: KRIEG DEM MÜLL)
|
||||
scratch*.py
|
||||
rewrite_*.py
|
||||
test_*.py
|
||||
!tests/**
|
||||
update_*.py
|
||||
profile_dump.*
|
||||
resp_dump.*
|
||||
test_compress.py
|
||||
test_fixtures.py
|
||||
output.txt
|
||||
|
||||
@@ -29,6 +29,13 @@ Found in `sensors/honeypot_radome.py`.
|
||||
- **Ghost Engagement Guard**: Strips DOM nodes explicitly tagged with `visible-to-user="false"` to prevent triggering Accessibility Hooks.
|
||||
- **VLM Sanity Guard**: Woven into `telepathic_engine.py`, it sends semantic matches for destructive actions (Like/Follow) through a Vision Language Model step to prevent executing semantic "Bait and Switch" tricks.
|
||||
|
||||
### 🧠 Situational Awareness Engine (SAE)
|
||||
Found in `situational_awareness.py`. Handles autonomous obstacle detection, recovery, and learning without hardcoded rules.
|
||||
- **3-Layer Modal Fast-Path**: Eliminates LLM hallucination traps for Instagram-internal modals (surveys, rating prompts) via O(1) deterministic structural checks:
|
||||
1. **Resource-ID Guard**: Detects internal blocking overlays (e.g., `survey_overlay_container`, `nux_overlay`).
|
||||
2. **Dismiss-Button Heuristic**: Cross-validates typical negative actions ("Not Now", "Take Survey") with overlay structures to prevent false positives in post captions.
|
||||
3. **Zero-Deception Fallback**: If structural markers fail, falls back to `ScreenMemoryDB` and ultimately the LLM. Structured invariants always override the semantic cache.
|
||||
|
||||
### 🦾 Biometric Facade (Gaussian Clicks)
|
||||
Found in `device_facade.py`.
|
||||
- Human touches do not follow a flat mathematical uniform grid. The GramPilot simulates genuine **biometric dispersion** using `random.gauss(mu, sigma)`, strictly centering clicks inside a thumb-bias radius (bottom-left skew for right-handers). In tests, this hits a 68% standard deviation precision.
|
||||
|
||||
@@ -47,7 +47,7 @@ def verify_and_switch_account(device, nav_graph, target_username):
|
||||
telepath = TelepathicEngine.get_instance()
|
||||
|
||||
# We ask the semantic engine to find the profile tab, ensuring 100% ID-agnostic behavior
|
||||
profile_tab_node = telepath.find_best_node(xml_dump, "tap profile tab", min_threshold=0.3)
|
||||
profile_tab_node = telepath.find_best_node(xml_dump, "tap profile tab", min_threshold=0.3, device=device)
|
||||
if profile_tab_node:
|
||||
profile_tab = (profile_tab_node["x"], profile_tab_node["y"])
|
||||
except Exception as e:
|
||||
@@ -113,7 +113,7 @@ def verify_and_switch_account(device, nav_graph, target_username):
|
||||
dump_ui_state(
|
||||
device, "identity_guard", {"reason": "account_not_found_in_bottom_sheet", "target": target_username}
|
||||
)
|
||||
except:
|
||||
except Exception:
|
||||
pass
|
||||
# Escape the bottom sheet
|
||||
device.press("back")
|
||||
|
||||
@@ -40,7 +40,15 @@ class DarwinDwellPlugin(BehaviorPlugin):
|
||||
logger.info("🐢 [DarwinDwell] Executing organic dwell behaviors...")
|
||||
darwin.execute_micro_wobble(ctx.device)
|
||||
res_score = ctx.shared_state.get("res_score", 1.0)
|
||||
darwin.execute_proof_of_resonance(ctx.device, res_score)
|
||||
darwin.execute_proof_of_resonance(
|
||||
ctx.device,
|
||||
res_score,
|
||||
nav_graph=ctx.cognitive_stack.get("nav_graph"),
|
||||
configs=ctx.configs,
|
||||
resonance_oracle=ctx.cognitive_stack.get("oracle"),
|
||||
username=ctx.username,
|
||||
context_xml=ctx.context_xml or ctx.device.dump_hierarchy(),
|
||||
)
|
||||
else:
|
||||
logger.info("🐢 [DarwinDwell] Darwin engine missing. Falling back to static sleep.")
|
||||
sleep(2.5 * ctx.sleep_mod)
|
||||
|
||||
@@ -49,7 +49,7 @@ class FollowPlugin(BehaviorPlugin):
|
||||
|
||||
nav_graph = QNavGraph(ctx.device)
|
||||
|
||||
if nav_graph.do("tap 'Follow' button"):
|
||||
if nav_graph.do("tap 'Follow' button") or nav_graph.do("tap 'Following' button"):
|
||||
logger.info(f"🤝 [Follow] Followed @{ctx.username} ✓")
|
||||
ctx.session_state.add_interaction(source=ctx.username, succeed=True, followed=True, scraped=False)
|
||||
|
||||
|
||||
@@ -42,6 +42,21 @@ class ObstacleGuardPlugin(BehaviorPlugin):
|
||||
|
||||
misses = ctx.shared_state.get("consecutive_marker_misses", 0)
|
||||
|
||||
# ── System Dialog / Permission Modal (e.g. "Allow Instagram to record audio?") ──
|
||||
if situation == SituationType.OBSTACLE_SYSTEM:
|
||||
logger.warning("⚠️ [ObstacleGuard] System permission dialog detected. Dismissing with BACK...")
|
||||
ctx.device.press("back")
|
||||
sleep(1.5 * ctx.sleep_mod)
|
||||
return BehaviorResult(executed=True, should_skip=True)
|
||||
|
||||
# ── Foreign App Takeover (e.g. browser opened, wrong app in foreground) ──
|
||||
if situation == SituationType.OBSTACLE_FOREIGN_APP:
|
||||
logger.warning("⚠️ [ObstacleGuard] Foreign app detected. Pressing BACK to recover...")
|
||||
ctx.device.press("back")
|
||||
sleep(1.5 * ctx.sleep_mod)
|
||||
return BehaviorResult(executed=True, should_skip=True)
|
||||
|
||||
# ── Instagram Modal / Overlay (survey, "Not Now" prompt, creation flow) ──
|
||||
if situation == SituationType.OBSTACLE_MODAL:
|
||||
if misses >= 2:
|
||||
logger.error("🛑 [ObstacleGuard] Failed to recover from OBSTACLE_MODAL after multiple attempts.")
|
||||
@@ -56,7 +71,7 @@ class ObstacleGuardPlugin(BehaviorPlugin):
|
||||
# Check recovery
|
||||
new_xml = ctx.device.dump_hierarchy()
|
||||
tele = TelepathicEngine.get_instance()
|
||||
best_node = tele.find_best_node(new_xml, intent_description="Dismiss obstacle")
|
||||
best_node = tele.find_best_node(new_xml, intent_description="Dismiss obstacle", device=ctx.device)
|
||||
if best_node:
|
||||
ctx.device.click(best_node.get("x", 0), best_node.get("y", 0))
|
||||
|
||||
|
||||
@@ -29,9 +29,11 @@ class PerfectSnappingPlugin(BehaviorPlugin):
|
||||
if not getattr(self, "_enabled", True):
|
||||
return False
|
||||
|
||||
xml_lower = ctx.context_xml.lower()
|
||||
# Do not snap if we are on a profile page or grid, it's meant for posts.
|
||||
if "profile_tabs_container" in xml_lower or "explore_grid" in xml_lower:
|
||||
# Perfect snapping is only for feed posts.
|
||||
# Do not snap if we are on a profile page, explore grid, or modal.
|
||||
from GramAddict.core.perception.feed_analysis import has_feed_markers
|
||||
|
||||
if not has_feed_markers(ctx.context_xml):
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
@@ -26,15 +26,24 @@ class PostDataExtractionPlugin(BehaviorPlugin):
|
||||
return 85
|
||||
|
||||
def can_activate(self, ctx: BehaviorContext) -> bool:
|
||||
return getattr(self, "_enabled", True) and ctx.context_xml is not None
|
||||
from GramAddict.core.perception.feed_analysis import has_feed_markers
|
||||
|
||||
return getattr(self, "_enabled", True) and ctx.context_xml is not None and has_feed_markers(ctx.context_xml)
|
||||
|
||||
def execute(self, ctx: BehaviorContext) -> BehaviorResult:
|
||||
logger.debug("🧩 [PostDataExtraction] Extracting post metadata...")
|
||||
post_data = extract_post_content(ctx.context_xml)
|
||||
post_data = extract_post_content(ctx.context_xml, device=ctx.device)
|
||||
|
||||
if post_data:
|
||||
ctx.post_data = post_data
|
||||
ctx.username = post_data.get("username", "")
|
||||
|
||||
if post_data.get("username_missing") or not ctx.username:
|
||||
logger.error(
|
||||
"❌ [PostDataExtraction] FAILED: Post author username is empty or missing! Halting interaction."
|
||||
)
|
||||
return BehaviorResult(executed=False, metadata={"error": "Empty username extracted"})
|
||||
|
||||
logger.info(f"📝 [PostDataExtraction] Post by @{ctx.username} extracted.")
|
||||
return BehaviorResult(executed=True)
|
||||
|
||||
|
||||
@@ -50,8 +50,11 @@ class RepostPlugin(BehaviorPlugin):
|
||||
|
||||
nav_graph = QNavGraph(ctx.device)
|
||||
|
||||
if nav_graph.do("share to story"):
|
||||
logger.info(f"📤 [Repost] Shared post by @{ctx.username} to story ✓")
|
||||
return BehaviorResult(executed=True, interactions=1)
|
||||
# We must click the send post button first
|
||||
if nav_graph.do("tap send post button"):
|
||||
# A modal should appear, now click add to story
|
||||
if nav_graph.do("tap add to story"):
|
||||
logger.info(f"📤 [Repost] Shared post by @{ctx.username} to story ✓")
|
||||
return BehaviorResult(executed=True, interactions=1)
|
||||
|
||||
return BehaviorResult(executed=False)
|
||||
|
||||
@@ -49,12 +49,36 @@ class ResonanceEvaluatorPlugin(BehaviorPlugin):
|
||||
tele = ctx.cognitive_stack.get("telepathic")
|
||||
if tele:
|
||||
logger.info("✨ [Resonance] Performing visual vibe check...")
|
||||
persona_interests = getattr(ctx.configs.args, "persona_interests", [])
|
||||
|
||||
# BUG 5 Fix: Read target_audience or persona_interests
|
||||
raw_interests = getattr(ctx.configs.args, "persona_interests", "")
|
||||
if not raw_interests:
|
||||
raw_interests = getattr(ctx.configs.args, "target_audience", "")
|
||||
|
||||
if isinstance(raw_interests, list):
|
||||
persona_interests = [str(i).strip() for i in raw_interests if str(i).strip()]
|
||||
else:
|
||||
persona_interests = [i.strip() for i in str(raw_interests).split(",") if i.strip()]
|
||||
|
||||
vibe = tele.evaluate_post_vibe(ctx.device, persona_interests)
|
||||
vibe_score = vibe.get("quality_score", 5) / 10.0
|
||||
if vibe.get("matches_niche"):
|
||||
vibe_score = min(1.0, vibe_score + 0.2)
|
||||
res_score = (res_score * 0.3) + (vibe_score * 0.7)
|
||||
if vibe is None:
|
||||
logger.warning(
|
||||
"✨ [Resonance] VLM vibe check returned None (truncated JSON?). Keeping neutral score."
|
||||
)
|
||||
else:
|
||||
if vibe.get("is_ad"):
|
||||
logger.info("🛡️ [Resonance Oracle] Visually identified post as an Ad! Skipping...")
|
||||
marker = vibe.get("ad_marker_text")
|
||||
if marker and marker.strip():
|
||||
from GramAddict.core.utils import learn_ad_marker
|
||||
learn_ad_marker(marker, ctx.context_xml)
|
||||
humanized_scroll(ctx.device)
|
||||
return BehaviorResult(executed=True, should_skip=True)
|
||||
|
||||
# BUG 6 Fix: VLM returns {"should_like": true/false}, not "quality_score"
|
||||
should_like = vibe.get("should_like", False)
|
||||
vibe_score = 1.0 if should_like else 0.2
|
||||
res_score = (res_score * 0.3) + (vibe_score * 0.7)
|
||||
|
||||
ctx.shared_state["res_score"] = res_score
|
||||
logger.info(f"📊 [Resonance] Post Score: {res_score:.2f}")
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import logging
|
||||
import os
|
||||
import random
|
||||
import re
|
||||
|
||||
try:
|
||||
import psutil
|
||||
@@ -92,9 +93,8 @@ def check_production_integrity():
|
||||
"""
|
||||
import sys
|
||||
|
||||
# If we are in a pytest session, we expect and allow mocks
|
||||
if "pytest" in sys.modules or "PYTEST_CURRENT_TEST" in os.environ:
|
||||
return
|
||||
# We no longer skip this in tests. Production integrity must hold everywhere.
|
||||
pass
|
||||
|
||||
try:
|
||||
from unittest.mock import MagicMock
|
||||
@@ -176,6 +176,15 @@ def start_bot(**kwargs):
|
||||
)
|
||||
persona_interests = [p.strip() for p in persona_raw.split(",") if p.strip()] if persona_raw else []
|
||||
|
||||
global_goal = getattr(configs.args, "goal", None)
|
||||
if global_goal:
|
||||
persona_interests.insert(0, global_goal)
|
||||
logger.info(
|
||||
f"🎯 [Autonomous Directive] Overriding target audience with high-level goal: {global_goal}",
|
||||
extra={"color": f"{Style.BRIGHT}{Fore.GREEN}"},
|
||||
)
|
||||
|
||||
from GramAddict.core.goap import GoalExecutor
|
||||
from GramAddict.core.interaction import LLMWriter
|
||||
from GramAddict.core.qdrant_memory import DMMemoryDB, ParasocialCRMDB
|
||||
from GramAddict.core.resonance_engine import ResonanceEngine
|
||||
@@ -188,7 +197,6 @@ def start_bot(**kwargs):
|
||||
active_inference = ActiveInferenceEngine(username)
|
||||
|
||||
# Core Autonomous Engines
|
||||
from GramAddict.core.goap import GoalExecutor
|
||||
|
||||
GoalExecutor.get_instance(device, username)
|
||||
zero_engine = ZeroLatencyEngine(device)
|
||||
@@ -298,7 +306,26 @@ def start_bot(**kwargs):
|
||||
cognitive_stack["dojo"] = dojo
|
||||
|
||||
try:
|
||||
bot_start_time = datetime.now()
|
||||
max_runtime = getattr(configs.args, "max_runtime_minutes", None)
|
||||
|
||||
dopamine.global_start_time = bot_start_time
|
||||
dopamine.global_max_runtime_minutes = max_runtime
|
||||
GoalExecutor.global_start_time = bot_start_time
|
||||
GoalExecutor.global_max_runtime_minutes = max_runtime
|
||||
|
||||
while True:
|
||||
if max_runtime:
|
||||
from datetime import timedelta
|
||||
|
||||
elapsed = datetime.now() - bot_start_time
|
||||
if elapsed > timedelta(minutes=max_runtime):
|
||||
logger.info(
|
||||
f"🛑 [Timeout] Maximum runtime of {max_runtime} minutes reached. Stopping bot.",
|
||||
extra={"color": f"{Fore.RED}"},
|
||||
)
|
||||
break
|
||||
|
||||
set_time_delta(configs.args)
|
||||
inside_working_hours, time_left = SessionState.inside_working_hours(
|
||||
configs.args.working_hours, configs.args.time_delta_session
|
||||
@@ -349,9 +376,7 @@ def start_bot(**kwargs):
|
||||
logger.info(
|
||||
f"🧠 [Agent Orchestrator] Session started. Strategy: {growth_brain.strategy} | Persona: {getattr(configs.args, 'agent_persona', 'unknown')}"
|
||||
)
|
||||
|
||||
from GramAddict.core.goap import GoalExecutor
|
||||
|
||||
# 1. Starten wir den GOAP Executor, um die UI-Struktur autonom zu erfassen
|
||||
goap = GoalExecutor.get_instance(device, username)
|
||||
|
||||
# --- PHASE 0: Autonomous Profile Scanning ---
|
||||
@@ -447,35 +472,68 @@ def start_bot(**kwargs):
|
||||
has_scanned_own_profile = True
|
||||
|
||||
while not dopamine.is_app_session_over():
|
||||
# 1. Ask the Growth Brain for a Desire
|
||||
current_desire = growth_brain.get_current_desire(dopamine)
|
||||
# ── 1. Generate available tasks from mission + plugins ──
|
||||
from GramAddict.core.goal_decomposer import GoalDecomposer
|
||||
|
||||
if current_desire == "ShiftContext":
|
||||
logger.info("🧠 [Free Will] Boredom critical. Forcing app restart to clear context.")
|
||||
device.app_stop(device.app_id)
|
||||
random_sleep(2.0, 4.0)
|
||||
device.app_start(device.app_id, use_monkey=True)
|
||||
random_sleep(4.0, 6.0)
|
||||
dopamine.boredom = max(0.0, dopamine.boredom * 0.2)
|
||||
continue
|
||||
decomposer = GoalDecomposer(
|
||||
plugins=configs.config.get("plugins", {}) if configs.config else {},
|
||||
actions={
|
||||
k: getattr(configs.args, k, None)
|
||||
for k in ("feed", "explore", "reels")
|
||||
if getattr(configs.args, k, None)
|
||||
},
|
||||
mission=configs.config.get("mission", {}) if configs.config else {},
|
||||
)
|
||||
available_tasks = decomposer.generate_tasks()
|
||||
|
||||
# 2. Map Desire to Sub-Feed
|
||||
target_map = {
|
||||
"DiscoverNewContent": ["ExploreFeed", "ReelsFeed"],
|
||||
"NurtureCommunity": ["HomeFeed", "StoriesFeed"],
|
||||
"SocialReciprocity": ["FollowingList"],
|
||||
}
|
||||
if not available_tasks:
|
||||
# No plugins enabled = nothing to do. Fall back to legacy desire system.
|
||||
current_desire = growth_brain.get_current_desire(dopamine)
|
||||
if current_desire == "ShiftContext":
|
||||
logger.info("🧠 [Free Will] Boredom critical. Forcing app restart.")
|
||||
device.app_stop(device.app_id)
|
||||
random_sleep(2.0, 4.0)
|
||||
device.app_start(device.app_id, use_monkey=True)
|
||||
random_sleep(4.0, 6.0)
|
||||
dopamine.boredom = max(0.0, dopamine.boredom * 0.2)
|
||||
continue
|
||||
|
||||
dm_config = configs.get_plugin_config("dm_reply")
|
||||
if dm_config.get("enabled", False):
|
||||
target_map["SocialReciprocity"].append("MessageInbox")
|
||||
# Legacy desire → target mapping (kept for backward compatibility)
|
||||
target_map = {
|
||||
"DiscoverNewContent": ["ExploreFeed", "ReelsFeed"],
|
||||
"NurtureCommunity": ["HomeFeed", "StoriesFeed"],
|
||||
"SocialReciprocity": ["FollowingList"],
|
||||
}
|
||||
|
||||
import secrets
|
||||
dm_config = configs.get_plugin_config("dm_reply")
|
||||
if dm_config.get("enabled", False):
|
||||
target_map["SocialReciprocity"].append("MessageInbox")
|
||||
|
||||
options = target_map.get(current_desire, ["HomeFeed"])
|
||||
current_target = secrets.choice(options)
|
||||
import secrets
|
||||
|
||||
logger.info(f"🧠 [Agent Orchestrator] Desire '{current_desire}' -> Routed to {current_target}")
|
||||
options = target_map.get(current_desire, ["HomeFeed"])
|
||||
current_target = secrets.choice(options)
|
||||
else:
|
||||
# ── 2. Select a concrete Task ──
|
||||
selected_task = growth_brain.select_task(dopamine, available_tasks)
|
||||
|
||||
if selected_task is None:
|
||||
# ShiftContext signal from high boredom
|
||||
logger.info("🧠 [Free Will] Boredom critical. Forcing app restart to clear context.")
|
||||
device.app_stop(device.app_id)
|
||||
random_sleep(2.0, 4.0)
|
||||
device.app_start(device.app_id, use_monkey=True)
|
||||
random_sleep(4.0, 6.0)
|
||||
dopamine.boredom = max(0.0, dopamine.boredom * 0.2)
|
||||
continue
|
||||
|
||||
current_target = selected_task.target_screen
|
||||
logger.info(
|
||||
f"🎯 [GoalDecomposer] Task: {selected_task.intent} "
|
||||
f"→ {current_target} (budget={selected_task.budget_posts})"
|
||||
)
|
||||
|
||||
logger.info(f"🧠 [Agent Orchestrator] Routed to {current_target}")
|
||||
|
||||
logger.info(f"⚡ Navigating to {current_target}")
|
||||
success = nav_graph.navigate_to(current_target, zero_engine)
|
||||
@@ -500,7 +558,9 @@ def start_bot(**kwargs):
|
||||
continue
|
||||
elif current_target == "StoriesFeed":
|
||||
logger.info("📱 Locating story tray on HomeFeed...")
|
||||
nav_graph.do("tap story ring avatar")
|
||||
if not nav_graph.do("tap story ring avatar"):
|
||||
logger.warning("❌ Failed to tap story ring avatar. Retrying next loop.")
|
||||
continue
|
||||
post_loaded = _wait_for_story_loaded(device, timeout=5)
|
||||
if not post_loaded:
|
||||
logger.warning("❌ Stories failed to open from HomeFeed. Retrying next loop.")
|
||||
@@ -625,6 +685,7 @@ def _interact_with_profile(device, configs, username, session_state, sleep_mod,
|
||||
|
||||
if cognitive_stack is None:
|
||||
cognitive_stack = {}
|
||||
_validate_cognitive_stack(cognitive_stack, "ProfileInteraction")
|
||||
|
||||
if hasattr(session_state, "my_username") and username == session_state.my_username:
|
||||
logger.info(f"🤝 [Deep Interaction] Skipping own profile @{username} to prevent self-interactions.")
|
||||
@@ -764,6 +825,7 @@ def _run_zero_latency_stories_loop(device, configs, session_state, cognitive_sta
|
||||
|
||||
from colorama import Fore
|
||||
|
||||
_validate_cognitive_stack(cognitive_stack, "StoriesLoop")
|
||||
logger.info("🎬 [StoriesFeed] Starting native story binging loop...", extra={"color": f"{Fore.CYAN}"})
|
||||
|
||||
dopamine = cognitive_stack.get("dopamine")
|
||||
@@ -798,6 +860,20 @@ def _run_zero_latency_stories_loop(device, configs, session_state, cognitive_sta
|
||||
logger.warning("Failed to dump UI hierarchy in StoriesFeed.")
|
||||
return "CONTEXT_LOST"
|
||||
|
||||
# ── Perimeter Guard: Verify we're still inside Instagram ──
|
||||
# Production bug 2026-05-03: A story's swipe-up link opened the Play Store,
|
||||
# and the loop kept tapping blindly on com.android.vending for 5+ iterations.
|
||||
packages = set(re.findall(r'package="([^"]+)"', xml_dump))
|
||||
app_id = getattr(device, "app_id", "com.instagram.android")
|
||||
if packages and app_id not in packages:
|
||||
logger.error(
|
||||
f"🚨 [StoriesFeed] FOREIGN APP DETECTED! Packages: {packages}. "
|
||||
f"A story link likely opened an external app. Aborting loop."
|
||||
)
|
||||
device.press("back")
|
||||
sleep(1.5)
|
||||
return "CONTEXT_LOST"
|
||||
|
||||
if getattr(configs.args, "ignore_close_friends", False):
|
||||
if "enge freunde" in xml_dump.lower() or "close friend" in xml_dump.lower():
|
||||
logger.info(
|
||||
@@ -819,6 +895,36 @@ def _run_zero_latency_stories_loop(device, configs, session_state, cognitive_sta
|
||||
return "FEED_EXHAUSTED"
|
||||
|
||||
|
||||
def _validate_cognitive_stack(cognitive_stack, context_name):
|
||||
"""
|
||||
Validates that the cognitive stack has all required engine dependencies
|
||||
injected properly. This is the ultimate zero-trust guard for plugins.
|
||||
"""
|
||||
if not isinstance(cognitive_stack, dict):
|
||||
raise TypeError(f"[{context_name}] CognitiveStack must be a dict, got {type(cognitive_stack)}")
|
||||
|
||||
required_engines = [
|
||||
"dopamine",
|
||||
"darwin",
|
||||
"resonance",
|
||||
"active_inference",
|
||||
"growth_brain",
|
||||
"swarm",
|
||||
"writer",
|
||||
"nav_graph",
|
||||
"zero_engine",
|
||||
"telepathic",
|
||||
]
|
||||
|
||||
missing = [eng for eng in required_engines if eng not in cognitive_stack or cognitive_stack[eng] is None]
|
||||
if missing:
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
logger.error(f"🚨 [{context_name}] CognitiveStack missing required engines: {missing}")
|
||||
raise ValueError(f"[{context_name}] CognitiveStack missing required engines: {missing}")
|
||||
|
||||
|
||||
def _run_zero_latency_feed_loop(
|
||||
device, zero_engine, nav_graph, configs, session_state, job_target, cognitive_stack, is_reels=False
|
||||
):
|
||||
@@ -832,6 +938,7 @@ def _run_zero_latency_feed_loop(
|
||||
- Darwin is the SOLE dwell controller → no duplicate sleep calls
|
||||
- SwarmProtocol emits pheromones after successful interactions
|
||||
"""
|
||||
_validate_cognitive_stack(cognitive_stack, "FeedLoop")
|
||||
logger.info(f"🔄 Entering Zero-Latency Interaction Pool. Feed: {job_target}")
|
||||
|
||||
dopamine = cognitive_stack.get("dopamine")
|
||||
@@ -867,6 +974,14 @@ def _run_zero_latency_feed_loop(
|
||||
|
||||
elif governance_decision == "CHECK_CURIOSITY":
|
||||
logger.info("👀 [Curiosity] Spontaneously checking DMs / Notifications...")
|
||||
|
||||
# 🛡️ Structural Guard: Curiosity targets (DMs, Notifications) are ONLY available on HomeFeed.
|
||||
# We must navigate there first, breaking current context.
|
||||
if not nav_graph.navigate_to("HomeFeed", zero_engine):
|
||||
logger.warning("❌ [Curiosity] Failed to navigate to HomeFeed. Aborting curiosity check.")
|
||||
continue
|
||||
sleep(random.uniform(1.0, 2.5))
|
||||
|
||||
dm_config = configs.get_plugin_config("dm_reply")
|
||||
if dm_config.get("enabled", False):
|
||||
explore_target = random.choice(["MessageInbox", "Notifications"])
|
||||
@@ -903,6 +1018,17 @@ def _run_zero_latency_feed_loop(
|
||||
if cognitive_stack.get("radome"):
|
||||
context_xml = cognitive_stack.get("radome").sanitize_xml(context_xml)
|
||||
|
||||
# ── Perimeter Guard: Verify we're still inside Instagram ──
|
||||
# Parity with story loop guard (production bug 2026-05-03).
|
||||
if context_xml:
|
||||
feed_packages = set(re.findall(r'package="([^"]+)"', context_xml))
|
||||
feed_app_id = getattr(device, "app_id", "com.instagram.android")
|
||||
if feed_packages and feed_app_id not in feed_packages:
|
||||
logger.error(f"🚨 [FeedLoop] FOREIGN APP DETECTED! Packages: {feed_packages}. Aborting loop.")
|
||||
device.press("back")
|
||||
sleep(1.5)
|
||||
return "CONTEXT_LOST"
|
||||
|
||||
# ── Execute Plugin Registry Behaviors (Feed Level) ──
|
||||
from GramAddict.core.behaviors import BehaviorContext, PluginRegistry
|
||||
|
||||
@@ -962,6 +1088,7 @@ def _run_zero_latency_search_loop(
|
||||
"""
|
||||
Executes the autonomous Search & Interact logic.
|
||||
"""
|
||||
_validate_cognitive_stack(cognitive_stack, "SearchLoop")
|
||||
logger.info("🧠 [Search Engine] Initiating keyword discovery...", extra={"color": f"{Style.BRIGHT}{Fore.CYAN}"})
|
||||
|
||||
import random
|
||||
@@ -987,6 +1114,16 @@ def _run_zero_latency_search_loop(
|
||||
xml = device.dump_hierarchy()
|
||||
telepathic = cognitive_stack.get("telepathic")
|
||||
|
||||
# ── Perimeter Guard: Verify we're still inside Instagram ──
|
||||
if xml:
|
||||
search_packages = set(re.findall(r'package="([^"]+)"', xml))
|
||||
search_app_id = getattr(device, "app_id", "com.instagram.android")
|
||||
if search_packages and search_app_id not in search_packages:
|
||||
logger.error(f"🚨 [SearchLoop] FOREIGN APP DETECTED! Packages: {search_packages}. Aborting loop.")
|
||||
device.press("back")
|
||||
sleep(1.5)
|
||||
return "CONTEXT_LOST"
|
||||
|
||||
# Find search bar
|
||||
search_bar = telepathic.find_best_node(xml, "Search edit text box or magnifying glass input", device=device)
|
||||
if search_bar:
|
||||
|
||||
@@ -17,13 +17,7 @@ class Config:
|
||||
self.args = kwargs
|
||||
self.module = True
|
||||
else:
|
||||
# Avoid parsing sys.argv if we are running in a test environment (pytest)
|
||||
# as pytest arguments will cause argparse to fail with SystemExit: 2
|
||||
is_pytest = "pytest" in sys.modules
|
||||
if is_pytest:
|
||||
self.args = []
|
||||
else:
|
||||
self.args = list(sys.argv)
|
||||
self.args = list(sys.argv)
|
||||
self.module = False
|
||||
|
||||
if not self.module and "--config" not in self.args:
|
||||
@@ -85,6 +79,9 @@ class Config:
|
||||
self.username = self.username[0]
|
||||
self.debug = self.config.get("debug", False)
|
||||
self.app_id = self.config.get("app_id", "com.instagram.android")
|
||||
|
||||
# Autonomous goals removed — the bot now derives tasks from mission + plugins
|
||||
# via GoalDecomposer. See GramAddict/core/goal_decomposer.py.
|
||||
else:
|
||||
if "--debug" in self.args:
|
||||
self.debug = True
|
||||
@@ -141,6 +138,9 @@ class Config:
|
||||
self.parser.add_argument("--total-sessions", help="Total amount of sessions", default="-1")
|
||||
self.parser.add_argument("--working-hours", help="Working hours", default=None)
|
||||
self.parser.add_argument("--time-delta-session", help="Time delta between sessions", default=None)
|
||||
self.parser.add_argument(
|
||||
"--max-runtime-minutes", type=int, help="Maximum runtime in minutes before bot auto-exits", default=None
|
||||
)
|
||||
self.parser.add_argument("--restart-atx-agent", action="store_true", help="Restart atx agent")
|
||||
self.parser.add_argument("--allow-untested-ig-version", action="store_true", help="Allow untested IG version")
|
||||
self.parser.add_argument(
|
||||
@@ -149,6 +149,13 @@ class Config:
|
||||
help="Wipe all learned navigation and telepathic memories on boot to start 100%% blank.",
|
||||
)
|
||||
|
||||
self.parser.add_argument(
|
||||
"--goal",
|
||||
type=str,
|
||||
help="High-level autonomous goal for the bot (Tesla-style). Overrides config.yml goals.",
|
||||
default=None,
|
||||
)
|
||||
|
||||
# Interaction settings
|
||||
self.parser.add_argument("--likes-count", help="Likes count", default="2-3")
|
||||
self.parser.add_argument("--likes-percentage", help="Likes percentage", default="100")
|
||||
|
||||
@@ -84,7 +84,6 @@ class DarwinEngine(QdrantBase):
|
||||
resonance: float,
|
||||
text_length: int = 0,
|
||||
nav_graph=None,
|
||||
zero_engine=None,
|
||||
configs=None,
|
||||
resonance_oracle=None,
|
||||
username=None,
|
||||
@@ -142,12 +141,22 @@ class DarwinEngine(QdrantBase):
|
||||
cy = h // 2
|
||||
|
||||
dur_ms = int(random.uniform(200, 500))
|
||||
device.shell(f"input swipe {int(cx)} {int(cy)} {int(cx + noise_x)} {int(cy + slip_distance)} {dur_ms}")
|
||||
|
||||
# Use physics-based injector instead of algorithmic 'input swipe'
|
||||
body = PhysicsBody.get_session_instance(device)
|
||||
injector = SendEventInjector.get_instance(device)
|
||||
start_pt = (int(cx), int(cy))
|
||||
end_pt = (int(cx + noise_x), int(cy + slip_distance))
|
||||
|
||||
points = BezierGesture.scroll_curve(start_pt, end_pt, body, n_points=5)
|
||||
timing = BezierGesture.compute_sigmoid_timing(len(points), dur_ms)
|
||||
injector.inject_gesture(points, timing, touch_major=body.get_touch_major())
|
||||
|
||||
time.sleep(random.uniform(0.5, 1.2))
|
||||
|
||||
# 4. Comment depth simulation (probabilistic & resonance-correlated)
|
||||
if profile["comment_read_dwell"] > 1.0 and resonance > 0.4 and random.random() < 0.3:
|
||||
if nav_graph and zero_engine:
|
||||
if nav_graph:
|
||||
if not self._has_comments(context_xml):
|
||||
logger.debug(" -> 🚫 [Darwin Engine] Skipping comment depth simulation (Post has 0 comments).")
|
||||
else:
|
||||
@@ -334,27 +343,34 @@ class DarwinEngine(QdrantBase):
|
||||
"""
|
||||
Heuristic to check if a post actually has comments to read.
|
||||
If it has 0 comments, checking them is suspicious bot behavior.
|
||||
|
||||
Zero-Maintenance: Uses only English text and resource_id patterns.
|
||||
Resource IDs are locale-invariant. English text in content_desc
|
||||
is used by Instagram internally and is reliable.
|
||||
"""
|
||||
low_xml = xml_string.lower()
|
||||
|
||||
# 1. Explicit zero comments checks
|
||||
if re.search(r"\b0\s*kommentare?\b", low_xml) or re.search(r"\b0\s*comment(?:s)?\b", low_xml):
|
||||
# 1. Explicit zero comments check (resource_id based + English fallback)
|
||||
if re.search(r"\b0\s*comment(?:s)?\b", low_xml):
|
||||
return False
|
||||
|
||||
# 2. Check for "view all" or similar prominent comment link texts
|
||||
if "view all" in low_xml or ("alle " in low_xml and "kommentare ansehen" in low_xml):
|
||||
if "view all" in low_xml:
|
||||
return True
|
||||
if "view 1 comment" in low_xml or "1 kommentar ansehen" in low_xml:
|
||||
if "view 1 comment" in low_xml:
|
||||
return True
|
||||
if "comment number is" in low_xml:
|
||||
return True
|
||||
|
||||
# 3. Check for specific counter elements > 0 in content descriptors
|
||||
# e.g. "by username, 23 comments" or "1,234 comments"
|
||||
has_number_of_comments = re.search(r"\b([1-9][0-9.,]*)\s*(?:comment(?:s)?|kommentare?)\b", low_xml)
|
||||
# 3. Structural: comment_textview_layout is present with a count > 0
|
||||
has_number_of_comments = re.search(r"\b([1-9][0-9.,]*)\s*comment(?:s)?\b", low_xml)
|
||||
if has_number_of_comments:
|
||||
return True
|
||||
|
||||
# 4. Structural: The comment button resource_id exists and has content
|
||||
if "row_feed_comment_textview_layout" in low_xml:
|
||||
return True
|
||||
|
||||
# If no indicators are found, assume the post has 0 comments.
|
||||
# The comment button exists, but there are no comments to read.
|
||||
return False
|
||||
|
||||
|
||||
@@ -36,15 +36,38 @@ def create_device(device_id, app_id, args=None):
|
||||
try:
|
||||
return DeviceFacade(device_id, app_id, args)
|
||||
except Exception as e:
|
||||
str(e)
|
||||
err_msg = str(e)
|
||||
err_type = str(type(e))
|
||||
if (
|
||||
"ConnectError" in err_type
|
||||
or "ConnectionRefusedError" in err_type
|
||||
or "ConnectionError" in err_type
|
||||
or "Timeout" in err_type
|
||||
if any(
|
||||
keyword in err_type or keyword in err_msg
|
||||
for keyword in ["ConnectError", "ConnectionRefused", "ConnectionError", "Timeout"]
|
||||
):
|
||||
logger.error(f"⚠️ [ADB ConnectError] Could not connect to device '{device_id}'.")
|
||||
|
||||
# Proactive Discovery
|
||||
try:
|
||||
import subprocess
|
||||
|
||||
result = subprocess.run(["adb", "devices"], capture_output=True, text=True, timeout=2)
|
||||
lines = [
|
||||
line.strip()
|
||||
for line in result.stdout.split("\n")
|
||||
if line.strip() and not line.startswith("List of devices")
|
||||
]
|
||||
devices = [line.split("\t")[0] for line in lines if "device" in line]
|
||||
|
||||
if devices:
|
||||
logger.info("🔍 Proactive Discovery: I found the following devices connected:")
|
||||
for d in devices:
|
||||
if d.split(":")[0] == device_id.split(":")[0]:
|
||||
logger.info(f" 👉 {d} (MATCHING IP - Is this the same device with a different port?)")
|
||||
else:
|
||||
logger.info(f" - {d}")
|
||||
else:
|
||||
logger.warning("🔍 Proactive Discovery: No ADB devices found. Is your phone authorized?")
|
||||
except Exception as discovery_err:
|
||||
logger.debug(f"Proactive discovery failed: {discovery_err}")
|
||||
|
||||
logger.error("👉 Please verify:")
|
||||
logger.error(" 1. Your phone is connected via USB or Wi-Fi.")
|
||||
logger.error(" 2. 'USB Debugging' is enabled in Developer Options.")
|
||||
@@ -359,6 +382,8 @@ class DeviceFacade:
|
||||
from io import BytesIO
|
||||
|
||||
img = self.deviceV2.screenshot()
|
||||
if img is None:
|
||||
return None
|
||||
buffered = BytesIO()
|
||||
img.save(buffered, format="JPEG", quality=70) # Compressed for target latency
|
||||
return base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
|
||||
@@ -13,20 +13,20 @@ MAX_REPLIES_PER_INBOX_VISIT = 3
|
||||
# Sentinel values that indicate missing message context.
|
||||
_EMPTY_CONTEXT_SENTINELS = frozenset({"no previous context", "", "none", "n/a"})
|
||||
|
||||
|
||||
# Structural resource-IDs that indicate a real "Send" button.
|
||||
_SEND_BUTTON_MARKERS = frozenset({"send_button", "row_thread_composer_send"})
|
||||
|
||||
|
||||
def _is_send_button(node: dict) -> bool:
|
||||
"""Structural verification: returns True only if the node is a real Send button."""
|
||||
attribs = node.get("original_attribs", {})
|
||||
rid = attribs.get("resource-id", "")
|
||||
desc = attribs.get("content-desc", node.get("desc", "")).lower()
|
||||
# Accept if resource-id contains a known send button marker
|
||||
if any(marker in rid for marker in _SEND_BUTTON_MARKERS):
|
||||
"""Semantic verification: returns True if the node is identified as a Send button."""
|
||||
desc = (node.get("description") or node.get("desc", "")).lower()
|
||||
text = (node.get("text") or "").lower()
|
||||
rid = (node.get("id") or node.get("resource_id", "")).lower()
|
||||
|
||||
# Accept if semantic markers indicate sending
|
||||
if any(m in rid for m in ["send", "composer_button"]):
|
||||
return True
|
||||
# Accept if content-desc is exactly "Send" (Instagram's canonical label)
|
||||
if desc == "send":
|
||||
if any(m in desc for m in ["send", "absenden"]):
|
||||
return True
|
||||
if text == "send" or text == "absenden":
|
||||
return True
|
||||
return False
|
||||
|
||||
@@ -83,16 +83,15 @@ def _run_zero_latency_dm_loop(device, zero_engine, nav_graph, configs, session_s
|
||||
xml_dump = device.dump_hierarchy()
|
||||
|
||||
# --- Zero Trust Structural Guard ---
|
||||
# -----------------------------------
|
||||
# ZERO TRUST STRUCTURAL GUARD
|
||||
# -----------------------------------
|
||||
# Validate we are actually in the Inbox or a Thread.
|
||||
# Hallucinations can lead to "Privacy Settings" or "Profile" screens.
|
||||
is_inbox = (
|
||||
'resource-id="com.instagram.android:id/inbox_refreshable_thread_list_recyclerview"' in xml_dump
|
||||
or 'resource-id="com.instagram.android:id/direct_inbox_action_bar"' in xml_dump
|
||||
)
|
||||
is_thread = 'resource-id="com.instagram.android:id/direct_thread_header"' in xml_dump
|
||||
from GramAddict.core.perception.screen_identity import ScreenIdentity, ScreenType
|
||||
|
||||
identity_engine = ScreenIdentity(getattr(configs.args, "username", ""))
|
||||
identity_engine.device = device
|
||||
screen_info = identity_engine.identify(xml_dump)
|
||||
|
||||
screen_type = screen_info["screen_type"]
|
||||
is_inbox = screen_type == ScreenType.DM_INBOX
|
||||
is_thread = screen_type == ScreenType.DM_THREAD
|
||||
|
||||
if is_thread:
|
||||
logger.warning("⚠️ [Structural Guard] DM Engine trapped in an open thread. Escaping...")
|
||||
@@ -102,9 +101,11 @@ def _run_zero_latency_dm_loop(device, zero_engine, nav_graph, configs, session_s
|
||||
sleep(1.5)
|
||||
continue
|
||||
|
||||
if not is_inbox and not is_thread:
|
||||
if not is_inbox:
|
||||
# We have drifted somewhere entirely alien (like Privacy Settings)
|
||||
logger.error("🛑 [Structural Guard] Alien context detected. Not in Inbox. Triggering CONTEXT_LOST.")
|
||||
logger.error(
|
||||
f"🛑 [Structural Guard] Alien context detected ({screen_type}). Not in Inbox. Triggering CONTEXT_LOST."
|
||||
)
|
||||
return "CONTEXT_LOST"
|
||||
# -----------------------------------
|
||||
|
||||
@@ -157,6 +158,7 @@ def _run_zero_latency_dm_loop(device, zero_engine, nav_graph, configs, session_s
|
||||
# Generate response
|
||||
prompt = f"You are replying to a direct message on Instagram. The last message you received was: '{context_text}'. Keep it short, casual, and friendly. Do not use hashtags."
|
||||
|
||||
logger.info(">>> [DM Engine] ABOUT TO CALL LLM")
|
||||
response_dict = query_llm(
|
||||
url=url,
|
||||
model=model,
|
||||
@@ -166,6 +168,7 @@ def _run_zero_latency_dm_loop(device, zero_engine, nav_graph, configs, session_s
|
||||
max_tokens=100,
|
||||
temperature=0.7,
|
||||
)
|
||||
logger.info(f">>> [DM Engine] LLM RETURNED: {response_dict}")
|
||||
|
||||
if response_dict and "response" in response_dict:
|
||||
response_text = response_dict["response"].strip()
|
||||
@@ -215,10 +218,13 @@ def _run_zero_latency_dm_loop(device, zero_engine, nav_graph, configs, session_s
|
||||
|
||||
# If keyboard was open, the first back only closed it. Check if still in thread.
|
||||
check_xml = device.dump_hierarchy()
|
||||
if (
|
||||
'resource-id="com.instagram.android:id/direct_thread_header"' in check_xml
|
||||
or 'resource-id="com.instagram.android:id/row_thread_composer_edittext"' in check_xml
|
||||
):
|
||||
from GramAddict.core.perception.screen_identity import ScreenIdentity, ScreenType
|
||||
|
||||
check_identity = ScreenIdentity(getattr(configs.args, "username", ""))
|
||||
check_identity.device = device
|
||||
check_screen = check_identity.identify(check_xml)
|
||||
|
||||
if check_screen["screen_type"] == ScreenType.DM_THREAD:
|
||||
device.press("back")
|
||||
sleep(1.0)
|
||||
|
||||
@@ -239,10 +245,13 @@ def _run_zero_latency_dm_loop(device, zero_engine, nav_graph, configs, session_s
|
||||
sleep(1.0)
|
||||
|
||||
check_xml = device.dump_hierarchy()
|
||||
if (
|
||||
'resource-id="com.instagram.android:id/direct_thread_header"' in check_xml
|
||||
or 'resource-id="com.instagram.android:id/row_thread_composer_edittext"' in check_xml
|
||||
):
|
||||
from GramAddict.core.perception.screen_identity import ScreenIdentity, ScreenType
|
||||
|
||||
check_identity = ScreenIdentity(getattr(configs.args, "username", ""))
|
||||
check_identity.device = device
|
||||
check_screen = check_identity.identify(check_xml)
|
||||
|
||||
if check_screen["screen_type"] == ScreenType.DM_THREAD:
|
||||
device.press("back")
|
||||
sleep(1.0)
|
||||
|
||||
|
||||
@@ -93,6 +93,18 @@ class DopamineEngine:
|
||||
)
|
||||
|
||||
def is_app_session_over(self):
|
||||
# Global Hard Kill check
|
||||
if getattr(self, "global_max_runtime_minutes", None):
|
||||
if hasattr(self, "global_start_time"):
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
if datetime.now() - self.global_start_time > timedelta(minutes=self.global_max_runtime_minutes):
|
||||
logger.info(
|
||||
f"🛑 [Timeout] Maximum runtime of {self.global_max_runtime_minutes} minutes reached (checked by DopamineEngine). Force-stopping session.",
|
||||
extra={"color": f"{Fore.RED}"},
|
||||
)
|
||||
return True
|
||||
|
||||
# True if we have scrolled too long or hit absolute burnout
|
||||
return (time.time() - self.session_start) > self.session_limit_seconds or self.boredom >= 100.0
|
||||
|
||||
|
||||
276
GramAddict/core/goal_decomposer.py
Normal file
276
GramAddict/core/goal_decomposer.py
Normal file
@@ -0,0 +1,276 @@
|
||||
"""
|
||||
GoalDecomposer — Mission-Driven Task Planning
|
||||
|
||||
Translates the bot's `mission` config + `plugins` capabilities into
|
||||
concrete, weighted Task objects. Pure logic — no LLM, no device,
|
||||
no network, no side effects.
|
||||
|
||||
This is the bridge between:
|
||||
- "What does the user WANT?" (mission.strategy)
|
||||
- "What CAN the bot DO?" (enabled plugins + actions)
|
||||
- "What SHOULD it do NOW?" (weighted Task selection)
|
||||
|
||||
Tesla analogy: FSD doesn't have a "goal: drive safely" config.
|
||||
It derives behavior from destination + road rules + sensor capabilities.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import random
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ── Strategy Weight Tables ──
|
||||
# Each strategy defines relative weights for screen targets.
|
||||
# Higher weight = more likely to be selected by GrowthBrain.
|
||||
STRATEGY_WEIGHTS: Dict[str, Dict[str, float]] = {
|
||||
"aggressive_growth": {
|
||||
"HomeFeed": 0.15,
|
||||
"ExploreFeed": 0.45,
|
||||
"ReelsFeed": 0.15,
|
||||
"StoriesFeed": 0.10,
|
||||
"MessageInbox": 0.10,
|
||||
"FollowingList": 0.05,
|
||||
},
|
||||
"community_builder": {
|
||||
"HomeFeed": 0.40,
|
||||
"ExploreFeed": 0.10,
|
||||
"ReelsFeed": 0.05,
|
||||
"StoriesFeed": 0.25,
|
||||
"MessageInbox": 0.15,
|
||||
"FollowingList": 0.05,
|
||||
},
|
||||
"passive_learning": {
|
||||
"HomeFeed": 0.20,
|
||||
"ExploreFeed": 0.50,
|
||||
"ReelsFeed": 0.20,
|
||||
"StoriesFeed": 0.05,
|
||||
"MessageInbox": 0.00,
|
||||
"FollowingList": 0.05,
|
||||
},
|
||||
"stealth_lurker": {
|
||||
"HomeFeed": 0.35,
|
||||
"ExploreFeed": 0.25,
|
||||
"ReelsFeed": 0.15,
|
||||
"StoriesFeed": 0.15,
|
||||
"MessageInbox": 0.05,
|
||||
"FollowingList": 0.05,
|
||||
},
|
||||
}
|
||||
|
||||
# ── Plugin → Screen Mapping ──
|
||||
# Which plugins enable which screen targets.
|
||||
# A screen is only viable if at least one enabling plugin is active.
|
||||
# Some plugins work on MULTIPLE screens (likes work on home, explore, reels).
|
||||
PLUGIN_SCREENS_MAP: Dict[str, set] = {
|
||||
"likes": {"HomeFeed", "ExploreFeed", "ReelsFeed"},
|
||||
"comment": {"HomeFeed", "ExploreFeed"},
|
||||
"follow": {"HomeFeed", "ExploreFeed"},
|
||||
"repost": {"HomeFeed", "ExploreFeed"},
|
||||
"profile_visit": {"HomeFeed", "ExploreFeed"},
|
||||
"grid_like": {"HomeFeed"},
|
||||
"carousel_browsing": {"HomeFeed"},
|
||||
"rabbit_hole": {"HomeFeed", "ExploreFeed"},
|
||||
"story_view": {"StoriesFeed"},
|
||||
"dm_reply": {"MessageInbox"},
|
||||
}
|
||||
|
||||
# ── Action → Screen Mapping ──
|
||||
# The `actions:` config section maps directly to screens.
|
||||
ACTION_SCREEN_MAP: Dict[str, str] = {
|
||||
"feed": "HomeFeed",
|
||||
"explore": "ExploreFeed",
|
||||
"reels": "ReelsFeed",
|
||||
}
|
||||
|
||||
# ── Screen → Verb Mapping ──
|
||||
SCREEN_VERB_MAP: Dict[str, str] = {
|
||||
"HomeFeed": "browse_feed",
|
||||
"ExploreFeed": "browse_explore",
|
||||
"ReelsFeed": "browse_reels",
|
||||
"StoriesFeed": "view_stories",
|
||||
"MessageInbox": "check_messages",
|
||||
"FollowingList": "manage_following",
|
||||
}
|
||||
|
||||
# ── Screen → Human Intent ──
|
||||
SCREEN_INTENT_MAP: Dict[str, str] = {
|
||||
"HomeFeed": "Interact with posts in the home feed",
|
||||
"ExploreFeed": "Discover and engage with new content",
|
||||
"ReelsFeed": "Browse and interact with reels",
|
||||
"StoriesFeed": "View and react to stories",
|
||||
"MessageInbox": "Reply to unread direct messages",
|
||||
"FollowingList": "Review and manage following list",
|
||||
}
|
||||
|
||||
DEFAULT_BUDGET = 5
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Task:
|
||||
"""A concrete, executable unit of work for the bot.
|
||||
|
||||
Unlike abstract goals ("nurture community"), a Task has:
|
||||
- A specific screen to navigate to
|
||||
- A measurable budget (how many posts/items to process)
|
||||
- A weight for probabilistic selection
|
||||
- A human-readable intent for logging
|
||||
"""
|
||||
|
||||
verb: str
|
||||
target_screen: str
|
||||
intent: str
|
||||
budget_posts: int
|
||||
weight: float
|
||||
|
||||
|
||||
class GoalDecomposer:
|
||||
"""Translates mission + plugins → weighted Task list.
|
||||
|
||||
Pure logic, zero side effects. Call generate_tasks() to get
|
||||
the bot's action menu for the current session.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
plugins: Dict[str, dict],
|
||||
actions: Dict[str, str],
|
||||
mission: Dict[str, str],
|
||||
):
|
||||
self._plugins = plugins
|
||||
self._actions = actions
|
||||
self._strategy = mission.get("strategy", "aggressive_growth")
|
||||
|
||||
def generate_tasks(self) -> List[Task]:
|
||||
"""Generate weighted tasks from config.
|
||||
|
||||
Returns an empty list if no plugins are enabled —
|
||||
the bot literally has nothing to do.
|
||||
"""
|
||||
viable_screens = self._discover_viable_screens()
|
||||
if not viable_screens:
|
||||
return []
|
||||
|
||||
strategy_weights = STRATEGY_WEIGHTS.get(self._strategy, STRATEGY_WEIGHTS["aggressive_growth"])
|
||||
|
||||
tasks = []
|
||||
for screen in viable_screens:
|
||||
weight = strategy_weights.get(screen, 0.1)
|
||||
if weight <= 0:
|
||||
continue
|
||||
|
||||
budget = self._budget_for_screen(screen)
|
||||
verb = SCREEN_VERB_MAP.get(screen, "browse")
|
||||
intent = SCREEN_INTENT_MAP.get(screen, f"Interact on {screen}")
|
||||
|
||||
tasks.append(
|
||||
Task(
|
||||
verb=verb,
|
||||
target_screen=screen,
|
||||
intent=intent,
|
||||
budget_posts=budget,
|
||||
weight=weight,
|
||||
)
|
||||
)
|
||||
|
||||
return tasks
|
||||
|
||||
def _discover_viable_screens(self) -> set:
|
||||
"""Determine which screens the bot can meaningfully interact on.
|
||||
|
||||
A screen is viable if it has BOTH:
|
||||
1. A route (action config or plugin-implied), AND
|
||||
2. At least one active plugin that can DO something there.
|
||||
|
||||
Without an active plugin, navigating to a screen is pointless —
|
||||
the bot would just scroll with nothing to interact on.
|
||||
"""
|
||||
# 1. Collect screens with active plugins
|
||||
plugin_screens: set = set()
|
||||
for plugin_name, screens in PLUGIN_SCREENS_MAP.items():
|
||||
plugin_cfg = self._plugins.get(plugin_name, {})
|
||||
if not plugin_cfg:
|
||||
continue
|
||||
if not self._is_plugin_active(plugin_cfg):
|
||||
continue
|
||||
plugin_screens.update(screens)
|
||||
|
||||
# 2. Screens from actions are only viable if plugins exist for them
|
||||
action_screens: set = set()
|
||||
for action_key, screen in ACTION_SCREEN_MAP.items():
|
||||
if action_key in self._actions and self._actions[action_key]:
|
||||
action_screens.add(screen)
|
||||
|
||||
# 3. A screen must have plugin coverage to be viable
|
||||
# Action-enabled screens need at least one active plugin
|
||||
viable = action_screens & plugin_screens
|
||||
|
||||
# 4. Plugin-only screens (story_view, dm_reply) are viable
|
||||
# even without an explicit action config
|
||||
viable |= plugin_screens
|
||||
|
||||
return viable
|
||||
|
||||
def _is_plugin_active(self, plugin_cfg: dict) -> bool:
|
||||
"""Check if a plugin config represents an active plugin.
|
||||
|
||||
A plugin is active if:
|
||||
- It has `enabled: true` (explicit), OR
|
||||
- It has `percentage` > 0 (implicit enable), OR
|
||||
- It has any config keys and `enabled` is not explicitly False
|
||||
"""
|
||||
# Explicit disable
|
||||
if plugin_cfg.get("enabled") is False:
|
||||
return False
|
||||
|
||||
# Explicit enable
|
||||
if plugin_cfg.get("enabled") is True:
|
||||
return True
|
||||
|
||||
# Percentage-based: 0% means disabled
|
||||
pct = plugin_cfg.get("percentage")
|
||||
if pct is not None:
|
||||
try:
|
||||
return float(pct) > 0
|
||||
except (ValueError, TypeError):
|
||||
return False
|
||||
|
||||
# Has config keys but no explicit enabled/percentage = active
|
||||
return bool(plugin_cfg)
|
||||
|
||||
def _budget_for_screen(self, screen: str) -> int:
|
||||
"""Determine the post budget for a screen.
|
||||
|
||||
Reads from actions config (e.g. feed: "5-10") and parses
|
||||
the range string into a random integer within bounds.
|
||||
"""
|
||||
# Map screen back to action key
|
||||
reverse_map = {v: k for k, v in ACTION_SCREEN_MAP.items()}
|
||||
action_key = reverse_map.get(screen)
|
||||
|
||||
if action_key and action_key in self._actions:
|
||||
return _parse_range(self._actions[action_key])
|
||||
|
||||
# Special screens get fixed budgets from plugin config
|
||||
if screen == "StoriesFeed":
|
||||
story_cfg = self._plugins.get("story_view", {})
|
||||
count_str = story_cfg.get("count", "1-3")
|
||||
return _parse_range(str(count_str))
|
||||
|
||||
if screen == "MessageInbox":
|
||||
return DEFAULT_BUDGET
|
||||
|
||||
return DEFAULT_BUDGET
|
||||
|
||||
|
||||
def _parse_range(range_str: str) -> int:
|
||||
"""Parse a range string like '5-10' into a random int within bounds."""
|
||||
try:
|
||||
if "-" in str(range_str):
|
||||
parts = str(range_str).split("-")
|
||||
low, high = int(parts[0]), int(parts[1])
|
||||
return random.randint(low, high)
|
||||
return int(range_str)
|
||||
except (ValueError, IndexError):
|
||||
return DEFAULT_BUDGET
|
||||
@@ -17,11 +17,11 @@ import logging
|
||||
import time
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from GramAddict.core.utils import random_sleep
|
||||
from GramAddict.core.navigation.knowledge import NavigationKnowledge
|
||||
from GramAddict.core.navigation.path_memory import PathMemory
|
||||
from GramAddict.core.navigation.planner import GoalPlanner
|
||||
from GramAddict.core.perception.screen_identity import ScreenIdentity, ScreenType
|
||||
from GramAddict.core.utils import random_sleep
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -43,6 +43,8 @@ class GoalExecutor:
|
||||
"""
|
||||
|
||||
_instance = None
|
||||
global_start_time = None
|
||||
global_max_runtime_minutes = None
|
||||
|
||||
@classmethod
|
||||
def get_instance(cls, device=None, bot_username=""):
|
||||
@@ -61,6 +63,7 @@ class GoalExecutor:
|
||||
self.device = device
|
||||
self.username = bot_username
|
||||
self.screen_id = ScreenIdentity(bot_username)
|
||||
self.screen_id.device = device
|
||||
self.planner = GoalPlanner(bot_username)
|
||||
self.path_memory = PathMemory(bot_username)
|
||||
self.max_steps = 15 # Safety: never execute more than 15 steps
|
||||
@@ -117,10 +120,25 @@ class GoalExecutor:
|
||||
consecutive_back_presses = 0
|
||||
MAX_CONSECUTIVE_BACK = 3
|
||||
explored_nav_actions = set()
|
||||
visited_screens = set()
|
||||
for step_num in range(max_steps):
|
||||
# ── Global Hard Kill Check ──
|
||||
max_rt = GoalExecutor.global_max_runtime_minutes
|
||||
start_time = GoalExecutor.global_start_time
|
||||
if max_rt and start_time:
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
if datetime.now() - start_time > timedelta(minutes=max_rt):
|
||||
logger.error(
|
||||
f"🛑 [Timeout] Maximum runtime of {max_rt} minutes reached during GOAP execution. Hard stopping planner.",
|
||||
extra={"color": "\\033[31m"},
|
||||
)
|
||||
return False
|
||||
|
||||
# PERCEIVE
|
||||
screen = self.perceive()
|
||||
screen_type = screen["screen_type"]
|
||||
visited_screens.add(screen_type)
|
||||
|
||||
if last_screen_type and screen_type != last_screen_type:
|
||||
logger.debug(
|
||||
@@ -134,7 +152,7 @@ class GoalExecutor:
|
||||
original_available = screen.get("available_actions", []).copy()
|
||||
masked_available = []
|
||||
for act in original_available:
|
||||
fail_count = self.action_failures.get(act, 0)
|
||||
fail_count = self.action_failures.get((screen_type, act), 0)
|
||||
if fail_count >= MAX_RETRIES:
|
||||
logger.warning(
|
||||
f"🚫 [GOAP] Masking action '{act}' due to {fail_count} consecutive failures to prevent loops."
|
||||
@@ -144,7 +162,7 @@ class GoalExecutor:
|
||||
screen["available_actions"] = masked_available
|
||||
|
||||
logger.debug(
|
||||
f"📍 [GOAP Step {step_num + 1}] On: {screen_type.value} | "
|
||||
f"📍 [GOAP Step {step_num + 1}] Goal: '{goal}' | On: {screen_type.value} | "
|
||||
f"Available: {screen.get('available_actions', [])[:5]}"
|
||||
)
|
||||
|
||||
@@ -156,13 +174,23 @@ class GoalExecutor:
|
||||
# SAE Feedback Loop!
|
||||
# If we hit this, the LAST action caused an obstacle! Mask it!
|
||||
if last_action and last_screen_type:
|
||||
self.action_failures[last_action] = (
|
||||
self.action_failures.get(last_action, 0) + MAX_RETRIES
|
||||
) # Instantly mask it
|
||||
self.planner.knowledge.learn_trap(last_screen_type, last_action, f"caused_obstacle_{obstacle_name}")
|
||||
logger.warning(
|
||||
f"🛡️ [SAE Feedback] Action '{last_action}' caused an obstacle. Masking aggressively and learned trap."
|
||||
)
|
||||
self.action_failures[(last_screen_type, last_action)] = (
|
||||
self.action_failures.get((last_screen_type, last_action), 0) + MAX_RETRIES
|
||||
) # Instantly mask it for this session
|
||||
from GramAddict.core.screen_topology import ScreenTopology
|
||||
|
||||
if ScreenTopology.is_structural_action(last_screen_type, last_action):
|
||||
logger.warning(
|
||||
f"🛡️ [SAE Feedback] Structural action '{last_action}' caused an obstacle. "
|
||||
f"Masking for this session. (Never burned permanently)"
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
f"🛡️ [SAE Feedback] Content action '{last_action}' caused an obstacle. "
|
||||
f"Masking for this session to break loop, but preventing permanent Qdrant poisoning."
|
||||
)
|
||||
# We specifically DO NOT call self.planner.knowledge.learn_trap here anymore!
|
||||
# Burning dynamic actions like "tap follow button" permanently destroys the bot's capabilities across sessions.
|
||||
|
||||
if not self._get_sae().ensure_clear_screen():
|
||||
if screen_type == ScreenType.FOREIGN_APP:
|
||||
@@ -172,7 +200,11 @@ class GoalExecutor:
|
||||
|
||||
# PLAN
|
||||
action = self.planner.plan_next_step(
|
||||
goal, screen, explored_nav_actions=explored_nav_actions, action_failures=self.action_failures
|
||||
goal,
|
||||
screen,
|
||||
explored_nav_actions=explored_nav_actions,
|
||||
action_failures=self.action_failures,
|
||||
visited_screens=visited_screens,
|
||||
)
|
||||
|
||||
if action is None:
|
||||
@@ -193,11 +225,20 @@ class GoalExecutor:
|
||||
|
||||
if success:
|
||||
steps_taken.append({"action": action})
|
||||
|
||||
if action == "force start instagram":
|
||||
logger.info("🔄 [GOAP State] App restarted. Purging memory/traps to attempt fresh routing.")
|
||||
self.action_failures.clear()
|
||||
explored_nav_actions.clear()
|
||||
visited_screens.clear()
|
||||
consecutive_back_presses = 0
|
||||
continue
|
||||
|
||||
# Check if it was a navigation action (vs a goal action). If we are not on the required screen,
|
||||
# any action taken is essentially a navigation attempt.
|
||||
explored_nav_actions.add(action)
|
||||
# Reset failures for this action since it eventually succeeded
|
||||
self.action_failures[action] = 0
|
||||
self.action_failures[(screen_type, action)] = 0
|
||||
|
||||
if "scroll" in action.lower():
|
||||
logger.debug(
|
||||
@@ -209,50 +250,55 @@ class GoalExecutor:
|
||||
from GramAddict.core.screen_topology import ScreenTopology
|
||||
|
||||
keys_to_clear = [
|
||||
k for k in self.action_failures.keys() if ScreenTopology.is_structural_action(screen_type, k)
|
||||
k
|
||||
for k in self.action_failures.keys()
|
||||
if k[0] == screen_type and ScreenTopology.is_structural_action(screen_type, k[1])
|
||||
]
|
||||
for k in keys_to_clear:
|
||||
del self.action_failures[k]
|
||||
|
||||
# ── Back-Press Circuit Breaker ──
|
||||
# ── Back-Press Circuit Breaker → Escalation ──
|
||||
if action == "press back":
|
||||
consecutive_back_presses += 1
|
||||
if consecutive_back_presses >= MAX_CONSECUTIVE_BACK:
|
||||
logger.error(
|
||||
logger.warning(
|
||||
f"🛑 [GOAP] Back-pressed {MAX_CONSECUTIVE_BACK} times with no screen transition. "
|
||||
f"Aborting goal '{goal}' to prevent app exit."
|
||||
f"Escalating to force restart."
|
||||
)
|
||||
self.path_memory.learn_path(goal, start_screen, steps_taken, False)
|
||||
|
||||
# Phase 3 GREEN: Unlearn the trap path
|
||||
# Unlearn the trap path
|
||||
from GramAddict.core.qdrant_memory import NavigationMemoryDB
|
||||
|
||||
# We don't know the exact action that got us here easily without analyzing steps_taken,
|
||||
# but we can grab the first action taken from start_screen in this chain if available.
|
||||
if len(steps_taken) > consecutive_back_presses:
|
||||
last_real_action = steps_taken[-consecutive_back_presses - 1]["action"]
|
||||
logger.debug(
|
||||
f"[GOAP Unlearn] last_real_action={last_real_action}, " f"start_screen={start_screen}"
|
||||
)
|
||||
NavigationMemoryDB().unlearn_transition(start_screen, last_real_action)
|
||||
else:
|
||||
logger.debug(
|
||||
f"[GOAP Unlearn] No real action to unlearn. "
|
||||
f"steps={len(steps_taken)}, back_presses={consecutive_back_presses}"
|
||||
)
|
||||
|
||||
return False
|
||||
# ── ESCALATION: Force restart instead of aborting ──
|
||||
app_id = getattr(self.device, "app_id", "com.instagram.android")
|
||||
self.device.app_start(app_id, use_monkey=True)
|
||||
random_sleep(2.0, 3.5)
|
||||
steps_taken.append({"action": "force start instagram"})
|
||||
|
||||
logger.info("🔄 [GOAP Escalation] App restarted. Purging all failure state for fresh attempt.")
|
||||
self.action_failures.clear()
|
||||
explored_nav_actions.clear()
|
||||
visited_screens.clear()
|
||||
consecutive_back_presses = 0
|
||||
continue
|
||||
else:
|
||||
consecutive_back_presses = 0
|
||||
else:
|
||||
self.action_failures[action] = self.action_failures.get(action, 0) + 1
|
||||
self.action_failures[(screen_type, action)] = self.action_failures.get((screen_type, action), 0) + 1
|
||||
# Track failed actions in explored_nav_actions so the planner
|
||||
# knows NOT to return the same synthetic intent again.
|
||||
# Without this, synthetic intents (not in available_actions)
|
||||
# bypass the masking logic and loop forever.
|
||||
explored_nav_actions.add(action)
|
||||
|
||||
if self.action_failures[action] >= MAX_RETRIES:
|
||||
if self.action_failures[(screen_type, action)] >= MAX_RETRIES:
|
||||
# ── Topology Guard: Never poison structural HD Map actions ──
|
||||
from GramAddict.core.screen_topology import ScreenTopology
|
||||
|
||||
@@ -354,11 +400,22 @@ class GoalExecutor:
|
||||
pre_action_screen_type = pre_action_screen["screen_type"]
|
||||
|
||||
# Determine if this was a navigation or an interaction
|
||||
from GramAddict.core.screen_topology import ScreenTopology
|
||||
|
||||
is_navigation = any(k in action.lower() for k in ["tab", "open", "go to", "navigate", "following list"])
|
||||
if not is_navigation:
|
||||
is_navigation = ScreenTopology.is_structural_action(pre_action_screen_type, action)
|
||||
action_success = False
|
||||
ui_changed = post_xml != xml_dump
|
||||
|
||||
# ── UI Change Detection with Noise Threshold ──
|
||||
# Raw string diffs of < 50 bytes are noise (timestamps, whitespace, counters).
|
||||
# A real navigation changes the XML by hundreds/thousands of bytes.
|
||||
MIN_UI_CHANGE_BYTES = 50
|
||||
xml_delta = abs(len(post_xml) - len(xml_dump))
|
||||
ui_changed = post_xml != xml_dump and xml_delta >= MIN_UI_CHANGE_BYTES
|
||||
logger.debug(
|
||||
f"[GOAP Verify] ui_changed={ui_changed}, " f"xml_len_pre={len(xml_dump)}, xml_len_post={len(post_xml)}"
|
||||
f"[GOAP Verify] ui_changed={ui_changed}, "
|
||||
f"xml_len_pre={len(xml_dump)}, xml_len_post={len(post_xml)}, delta={xml_delta}b"
|
||||
)
|
||||
|
||||
if is_navigation:
|
||||
@@ -414,7 +471,16 @@ class GoalExecutor:
|
||||
action_success = False
|
||||
else:
|
||||
# For interactions (like, follow) or unknown goals, use XML delta + semantic verify
|
||||
if ui_changed:
|
||||
# REGRESSION FIX 2026-05-01: Toggle actions (like/save) produce tiny XML deltas
|
||||
# (e.g. checked="false" → "true" = 1 byte). We must NOT gate interactions on
|
||||
# MIN_UI_CHANGE_BYTES. ANY change at all warrants semantic verification.
|
||||
interaction_xml_changed = post_xml != xml_dump
|
||||
if post_screen_type == ScreenType.FOREIGN_APP:
|
||||
logger.error(
|
||||
f"❌ [GOAP Verify] Interaction '{action}' caused navigation to FOREIGN_APP (e.g. Play Store). Rejecting as catastrophic failure."
|
||||
)
|
||||
action_success = False
|
||||
elif interaction_xml_changed:
|
||||
score = best_node.get("score", 0.0) if best_node else 0.0
|
||||
verification = engine.verify_success(action, post_xml, device=self.device, confidence=score)
|
||||
if verification is True:
|
||||
@@ -447,8 +513,12 @@ class GoalExecutor:
|
||||
return False
|
||||
else:
|
||||
# action_success is None (INCONCLUSIVE)
|
||||
# We decay the memory so it unlearns if it repeatedly fails to produce a definitive success.
|
||||
logger.warning(f"⚠️ [GOAP Execute] Applying AGGRESSIVE PENALTY for inconclusive action '{action}'.")
|
||||
engine.decay_click(action)
|
||||
# Double penalty to burn ambiguous paths faster (outer loop adds +1, so total +2 = instantly hits MAX_RETRIES)
|
||||
self.action_failures[(pre_action_screen_type, action)] = (
|
||||
self.action_failures.get((pre_action_screen_type, action), 0) + 1
|
||||
)
|
||||
return False
|
||||
|
||||
def _execute_recalled_path(self, steps: List[Dict], goal: str) -> bool:
|
||||
|
||||
@@ -94,6 +94,63 @@ class GrowthBrain:
|
||||
logger.info(f"🧠 [GrowthBrain] Strategy '{self.strategy}' dictated Desire: {selected_desire}")
|
||||
return selected_desire
|
||||
|
||||
def get_current_goal(self, dopamine_engine, available_goals: list[str], success_rates: dict = None) -> str:
|
||||
"""
|
||||
Autonomously selects the next strategic goal.
|
||||
If no goals are configured, falls back to legacy desires.
|
||||
Weights goals based on session success rates if provided.
|
||||
|
||||
.. deprecated::
|
||||
Use select_task() instead for concrete, plugin-linked task selection.
|
||||
"""
|
||||
import random
|
||||
|
||||
if not available_goals:
|
||||
# Legacy Desire Mapping (Fallback)
|
||||
return self.get_current_desire(dopamine_engine)
|
||||
|
||||
if dopamine_engine.boredom > 80:
|
||||
return "ShiftContext" # High boredom triggers a context shift
|
||||
|
||||
if not success_rates:
|
||||
return random.choice(available_goals)
|
||||
|
||||
weights = []
|
||||
for goal in available_goals:
|
||||
base_weight = 1.0
|
||||
success_count = success_rates.get(goal, 0)
|
||||
weight = base_weight + float(success_count)
|
||||
weights.append(weight)
|
||||
|
||||
return random.choices(available_goals, weights=weights, k=1)[0]
|
||||
|
||||
def select_task(self, dopamine_engine, available_tasks: list) -> "Optional[Task]":
|
||||
"""Select the next concrete Task using weighted random selection.
|
||||
|
||||
This is the primary interface for the orchestrator. Unlike get_current_goal()
|
||||
which returns abstract strings, this returns a Task object with a specific
|
||||
target_screen, budget, and success metric.
|
||||
|
||||
Returns:
|
||||
Task: The selected task to execute.
|
||||
None: If no tasks available or boredom is too high (ShiftContext signal).
|
||||
"""
|
||||
if not available_tasks:
|
||||
return None
|
||||
|
||||
# High boredom = ShiftContext (take a break, switch feed)
|
||||
if dopamine_engine.boredom > 85.0:
|
||||
logger.info("🧠 [GrowthBrain] Boredom too high for task selection. ShiftContext.")
|
||||
return None
|
||||
|
||||
weights = [task.weight for task in available_tasks]
|
||||
selected = random.choices(available_tasks, weights=weights, k=1)[0]
|
||||
logger.info(
|
||||
f"🧠 [GrowthBrain] Selected task: {selected.verb} → {selected.target_screen} "
|
||||
f"(weight={selected.weight:.2f}, budget={selected.budget_posts})"
|
||||
)
|
||||
return selected
|
||||
|
||||
def get_circadian_pacing(self) -> float:
|
||||
"""
|
||||
Adjusts activity levels based on the current local time
|
||||
|
||||
@@ -287,6 +287,12 @@ def query_llm(
|
||||
req_data["images"] = images_b64
|
||||
if format_json:
|
||||
req_data["format"] = "json"
|
||||
else:
|
||||
# For free-text calls (Brain action extraction), explicitly disable
|
||||
# thinking mode. Reasoning models like qwen3.5 put EVERYTHING in
|
||||
# the thinking block and return response='', which is useless for
|
||||
# action extraction. think=false forces a direct response.
|
||||
req_data["think"] = False
|
||||
|
||||
# Ollama passes configs inside 'options'
|
||||
if temperature is not None or max_tokens is not None:
|
||||
@@ -344,16 +350,27 @@ def query_llm(
|
||||
return {"response": content}
|
||||
else:
|
||||
# Ollama returns response OR thinking (for reasoning models)
|
||||
content = resp_json.get("response") or resp_json.get("thinking") or ""
|
||||
raw_response = resp_json.get("response", "")
|
||||
raw_thinking = resp_json.get("thinking", "")
|
||||
|
||||
logger.debug(f"DEBUG LLM PAYLOAD: response='{raw_response}', thinking='{raw_thinking}'")
|
||||
|
||||
# CRITICAL: For free-text mode (format_json=False), do NOT substitute
|
||||
# thinking for empty response. The thinking block is REASONING, not
|
||||
# a decision. The Brain parser would extract random actions from it.
|
||||
# For JSON mode (format_json=True), falling back to thinking IS correct
|
||||
# because reasoning models may place structured output in the thinking block.
|
||||
if format_json:
|
||||
content = raw_response or raw_thinking or ""
|
||||
extracted = extract_json(content)
|
||||
if not extracted:
|
||||
# Log more context if JSON extraction fails
|
||||
logger.debug(f"Ollama raw content (for JSON extraction): {content[:200]}...")
|
||||
raise ValueError("Ollama returned non-JSON content when JSON was expected.")
|
||||
resp_json["response"] = extracted
|
||||
logger.warning(f"Failed to extract JSON from content: {content[:100]}")
|
||||
else:
|
||||
content = extracted
|
||||
else:
|
||||
content = raw_response
|
||||
|
||||
return resp_json
|
||||
return {"response": content}
|
||||
except requests.exceptions.ConnectionError:
|
||||
logger.error(f"⚠️ [LLM Provider] Connection refused for {model} at {url}. Is the service running?")
|
||||
except Exception as e:
|
||||
|
||||
@@ -15,7 +15,11 @@ def ask_brain_for_action(
|
||||
return None
|
||||
|
||||
cfg = Config()
|
||||
url = getattr(cfg.args, "ai_model_url", "http://localhost:11434/api/generate") if hasattr(cfg, "args") else "http://localhost:11434/api/generate"
|
||||
url = (
|
||||
getattr(cfg.args, "ai_model_url", "http://localhost:11434/api/generate")
|
||||
if hasattr(cfg, "args")
|
||||
else "http://localhost:11434/api/generate"
|
||||
)
|
||||
model = getattr(cfg.args, "ai_model", "qwen3.5:latest") if hasattr(cfg, "args") else "qwen3.5:latest"
|
||||
|
||||
prompt = (
|
||||
@@ -32,7 +36,7 @@ def ask_brain_for_action(
|
||||
"INSTRUCTIONS:\n"
|
||||
"1. Reason about where you are. Consider the screen type and what actions make sense on that screen.\n"
|
||||
"2. If the goal requires navigating away from the current screen, choose the action that moves you closest to the goal.\n"
|
||||
"3. 'scroll down' reveals more UI elements on scrollable screens (feeds, profiles, lists). Some screens like stories or modals are NOT scrollable.\n"
|
||||
"3. 'scroll down' reveals more UI elements on scrollable screens (feeds, profiles, lists). If your target is likely on this screen but not currently visible, you MUST choose 'scroll down'.\n"
|
||||
"4. 'press back' exits the current screen and returns to the previous one. Use it when you are on a screen that doesn't lead to your goal.\n"
|
||||
"5. DO NOT hallucinate actions. Reply ONLY with the exact string from the available actions list.\n"
|
||||
"6. Reply with ONLY the action string, nothing else."
|
||||
@@ -40,18 +44,43 @@ def ask_brain_for_action(
|
||||
|
||||
try:
|
||||
response = query_llm(
|
||||
url=url, model=model, prompt="Choose the next best action.", system=prompt, format_json=False
|
||||
url=url,
|
||||
model=model,
|
||||
prompt="Choose the next best action.",
|
||||
system=prompt,
|
||||
format_json=False,
|
||||
max_tokens=250,
|
||||
)
|
||||
if response:
|
||||
result = response if isinstance(response, str) else response.get("response", "")
|
||||
result = result.strip().strip("'\"")
|
||||
|
||||
# Fuzzy match to available actions just in case
|
||||
# 1. Exact match check (ideal case)
|
||||
for act in available_actions:
|
||||
if act.lower() in result.lower():
|
||||
if act.lower() == result.lower():
|
||||
return act
|
||||
|
||||
# 2. Strict line-by-line check (often the model outputs the action on the last line)
|
||||
for line in reversed(result.splitlines()):
|
||||
line = line.strip().strip("'\"")
|
||||
for act in available_actions:
|
||||
if act.lower() == line.lower():
|
||||
return act
|
||||
|
||||
logger.warning(f"🧠 [Brain] LLM returned an invalid action: '{result}'. Falling back.")
|
||||
# 3. Fuzzy match (find the LAST mentioned action in the text, assuming it's the conclusion)
|
||||
best_act = None
|
||||
best_idx = -1
|
||||
for act in available_actions:
|
||||
idx = result.lower().rfind(act.lower())
|
||||
if idx > best_idx:
|
||||
best_idx = idx
|
||||
best_act = act
|
||||
|
||||
if best_act:
|
||||
logger.warning(f"🧠 [Brain] Extracted action '{best_act}' from verbose LLM output.")
|
||||
return best_act
|
||||
|
||||
logger.warning(f"🧠 [Brain] LLM returned an invalid action or no action found: '{result[:100]}...'. Falling back.")
|
||||
except Exception as e:
|
||||
logger.debug(f"🧠 [Brain] Error querying LLM: {e}")
|
||||
|
||||
|
||||
@@ -109,7 +109,7 @@ class PathMemory:
|
||||
try:
|
||||
from qdrant_client import models
|
||||
|
||||
point_id = self._db._get_id(seed)
|
||||
point_id = self._db.generate_uuid(seed)
|
||||
self._db.client.delete(
|
||||
collection_name=self._db.collection_name, points_selector=models.PointIdsList(points=[point_id])
|
||||
)
|
||||
|
||||
@@ -18,7 +18,12 @@ class GoalPlanner:
|
||||
self.knowledge = NavigationKnowledge(username)
|
||||
|
||||
def plan_next_step(
|
||||
self, goal: str, screen: Dict[str, Any], explored_nav_actions: set = None, action_failures: dict = None
|
||||
self,
|
||||
goal: str,
|
||||
screen: Dict[str, Any],
|
||||
explored_nav_actions: set = None,
|
||||
action_failures: dict = None,
|
||||
visited_screens: set = None,
|
||||
) -> Optional[str]:
|
||||
"""Plans the NEXT single action to take toward the goal."""
|
||||
screen_type = screen["screen_type"]
|
||||
@@ -37,7 +42,7 @@ class GoalPlanner:
|
||||
# ── 3. Am I on the right screen? If not, navigate there ──
|
||||
selected_tab = screen.get("selected_tab")
|
||||
nav_action = self._plan_navigation(
|
||||
goal_lower, screen_type, available, selected_tab, explored_nav_actions, action_failures
|
||||
goal_lower, screen_type, available, selected_tab, explored_nav_actions, action_failures, visited_screens
|
||||
)
|
||||
if nav_action:
|
||||
return nav_action
|
||||
@@ -75,6 +80,7 @@ class GoalPlanner:
|
||||
selected_tab: Optional[str] = None,
|
||||
explored_nav_actions: set = None,
|
||||
action_failures: dict = None,
|
||||
visited_screens: set = None,
|
||||
) -> Optional[str]:
|
||||
"""If we're on the wrong screen, figure out how to navigate.
|
||||
|
||||
@@ -94,35 +100,64 @@ class GoalPlanner:
|
||||
logger.debug(f"🛡️ [Aversive Filter] Masking trapped action: '{action}'")
|
||||
available = safe_available
|
||||
|
||||
visited_screens = visited_screens or set()
|
||||
|
||||
# 0b. No-Op Guard & Anti-Loop Guard:
|
||||
# - Strip tab actions that navigate to the CURRENT screen.
|
||||
# - Strip actions that navigate to PREVIOUSLY VISITED screens (except back-tracking).
|
||||
noop_actions = set()
|
||||
for action in available:
|
||||
expected = ScreenTopology.expected_screen_for_action(action, screen_type)
|
||||
if expected == screen_type:
|
||||
noop_actions.add(action)
|
||||
logger.debug(f"🛡️ [No-Op Guard] Stripping '{action}' — leads back to {screen_type.name}")
|
||||
elif expected in visited_screens and action != "press back":
|
||||
noop_actions.add(action)
|
||||
logger.debug(f"🛡️ [Anti-Loop Guard] Stripping '{action}' — leads to visited {expected.name}")
|
||||
|
||||
# Also strip actions where the HD Map says they go TO the current screen from OTHER screens
|
||||
for src_screen, transitions in ScreenTopology.TRANSITIONS.items():
|
||||
if src_screen == screen_type:
|
||||
continue # We already handled this screen's own transitions
|
||||
for action, dest in transitions.items():
|
||||
if dest == screen_type and action in available:
|
||||
noop_actions.add(action)
|
||||
logger.debug(
|
||||
f"🛡️ [No-Op Guard] Stripping '{action}' — known to navigate to current {screen_type.name}"
|
||||
)
|
||||
elif dest in visited_screens and action in available and action != "press back":
|
||||
noop_actions.add(action)
|
||||
logger.debug(f"🛡️ [Anti-Loop Guard] Stripping '{action}' — known to navigate to visited {dest.name}")
|
||||
|
||||
available = [a for a in available if a not in noop_actions]
|
||||
|
||||
# Build avoid_actions for HD Map route planning
|
||||
avoid_actions = (explored_nav_actions or set()).copy()
|
||||
if action_failures:
|
||||
for act, count in action_failures.items():
|
||||
if count >= 2: # MAX_RETRIES is 2 in goap
|
||||
avoid_actions.add(act)
|
||||
for key, count in action_failures.items():
|
||||
if isinstance(key, tuple) and len(key) == 2:
|
||||
scr, act = key
|
||||
if scr == screen_type and count >= 2: # MAX_RETRIES is 2 in goap
|
||||
avoid_actions.add(act)
|
||||
else:
|
||||
if count >= 2:
|
||||
avoid_actions.add(key)
|
||||
|
||||
target_screen = ScreenTopology.goal_to_target_screen(goal)
|
||||
|
||||
|
||||
# ── 1. HD Map Pre-Check for Dead Ends ──
|
||||
# If the topological map KNOWS the target is unreachable due to action_failures,
|
||||
# we must preempt the Brain from blindly routing into a dead end.
|
||||
if target_screen and target_screen != screen_type:
|
||||
route = ScreenTopology.find_route(screen_type, target_screen, avoid_actions=avoid_actions)
|
||||
if route is None and ScreenTopology.find_route(screen_type, target_screen):
|
||||
logger.warning(f"🛡️ [HD Map] Target {target_screen.name} is unreachable due to masked edges! Preventing Brain from blind routing.")
|
||||
logger.warning(
|
||||
f"🛡️ [HD Map] Target {target_screen.name} is unreachable due to masked edges! Preventing Brain from blind routing."
|
||||
)
|
||||
return None
|
||||
|
||||
# ── 2. Brain-Driven Decision Making (Primary Strategy) ──
|
||||
# The user explicitly wants the AI to be the primary driver of goals.
|
||||
from GramAddict.core.navigation.brain import ask_brain_for_action
|
||||
|
||||
brain_action = ask_brain_for_action(goal, screen_type.name, available, avoid_actions)
|
||||
if brain_action:
|
||||
logger.info(f"🧠 [Brain] Decided dynamically to execute: '{brain_action}'")
|
||||
return brain_action
|
||||
|
||||
# ── 2. HD Map Routing (Fallback) ──
|
||||
# If the Brain doesn't know what to do, try the deterministic topological map.
|
||||
# ── 2. HD Map Routing (Primary Strategy for Navigation) ──
|
||||
# Ground UI transitions in structural invariants. If the topological map knows the route, use it.
|
||||
target_screen = ScreenTopology.goal_to_target_screen(goal)
|
||||
if target_screen and target_screen != screen_type:
|
||||
route = ScreenTopology.find_route(screen_type, target_screen, avoid_actions=avoid_actions)
|
||||
@@ -143,6 +178,15 @@ class GoalPlanner:
|
||||
f"🛡️ [HD Map] Route action '{next_action}' already explored and failed. Skipping HD Map."
|
||||
)
|
||||
|
||||
# ── 3. Brain-Driven Decision Making (Fallback / Discovery) ──
|
||||
# For non-navigation goals or when the HD Map is incomplete.
|
||||
from GramAddict.core.navigation.brain import ask_brain_for_action
|
||||
|
||||
brain_action = ask_brain_for_action(goal, screen_type.name, available, avoid_actions)
|
||||
if brain_action:
|
||||
logger.info(f"🧠 [Brain] Decided to execute: '{brain_action}' (to achieve: '{goal}')")
|
||||
return brain_action
|
||||
|
||||
# ── 2. Learned Knowledge (Qdrant) ──
|
||||
required_screens = self.knowledge.get_requirements(goal)
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import json
|
||||
import logging
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
@@ -5,6 +6,39 @@ from GramAddict.core.perception.spatial_parser import SpatialNode
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _parse_yes_no(response: str) -> Optional[bool]:
|
||||
"""Parses a VLM response to find a definitive YES or NO without substring-matching 'not' or 'now'."""
|
||||
text = response.strip()
|
||||
|
||||
# Try parsing as JSON first
|
||||
if text.startswith("{"):
|
||||
try:
|
||||
data = json.loads(text)
|
||||
for k, v in data.items():
|
||||
if str(k).strip().upper() == "YES" or str(v).strip().upper() == "YES":
|
||||
return True
|
||||
if str(k).strip().upper() == "NO" or str(v).strip().upper() == "NO":
|
||||
return False
|
||||
if str(k).strip().lower() == "success" and isinstance(v, bool):
|
||||
return v
|
||||
|
||||
# If it is valid JSON but we couldn't definitively find YES/NO,
|
||||
# do NOT fall through to text matching
|
||||
return None
|
||||
except Exception:
|
||||
# Prevent JSON parsing fall-throughs
|
||||
return None
|
||||
|
||||
text_lower = text.lower()
|
||||
if text_lower.startswith("yes"):
|
||||
return True
|
||||
if text_lower.startswith("no") and not text_lower.startswith("now") and not text_lower.startswith("not"):
|
||||
return False
|
||||
|
||||
return None
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════
|
||||
# Semantic Match Keywords — SSOT for intent → element validation
|
||||
# ═══════════════════════════════════════════════════════
|
||||
@@ -13,10 +47,12 @@ logger = logging.getLogger(__name__)
|
||||
# If the intent contains the key, the clicked element MUST
|
||||
# contain at least one of the corresponding markers in its
|
||||
# text, content_desc, or resource_id.
|
||||
# ZERO MAINTENANCE: Only English words and resource_id fragments allowed.
|
||||
# No localized strings — the bot must work on any device language.
|
||||
TOGGLE_INTENT_MARKERS = {
|
||||
"follow": ["follow", "gefolgt", "abonnieren"],
|
||||
"like": ["like", "heart", "gefällt"],
|
||||
"save": ["save", "saved", "bookmark", "speichern"],
|
||||
"follow": ["follow", "button_follow"],
|
||||
"like": ["like", "heart", "button_like"],
|
||||
"save": ["save", "saved", "bookmark"],
|
||||
}
|
||||
|
||||
|
||||
@@ -71,7 +107,10 @@ class ActionMemory:
|
||||
self._last_click_context = None
|
||||
return
|
||||
|
||||
logger.info(f"✅ [ActionMemory] Confirming success for '{ctx['intent']}'. Boosting confidence.")
|
||||
logger.info(
|
||||
f"✅ [ActionMemory] Confirming success for '{ctx['intent']}'. Boosting confidence.",
|
||||
extra={"color": "\x1b[32m"},
|
||||
)
|
||||
|
||||
# Store or boost in Qdrant
|
||||
try:
|
||||
@@ -95,7 +134,9 @@ class ActionMemory:
|
||||
if intent and ctx["intent"] != intent:
|
||||
return
|
||||
|
||||
logger.warning(f"❌ [ActionMemory] Click failed for '{ctx['intent']}'. Applying penalty.")
|
||||
logger.warning(
|
||||
f"❌ [ActionMemory] Click failed for '{ctx['intent']}'. Applying penalty.", extra={"color": "\x1b[31m"}
|
||||
)
|
||||
|
||||
try:
|
||||
self.ui_memory.decay_confidence(ctx["intent"], ctx["xml_context"])
|
||||
@@ -113,26 +154,49 @@ class ActionMemory:
|
||||
intent_lower = intent.lower()
|
||||
post_xml_lower = post_click_xml.lower()
|
||||
|
||||
# Specific check for explore grid
|
||||
if "first image in explore grid" in intent_lower or "grid item" in intent_lower:
|
||||
if "row_feed_photo_imageview" in post_xml_lower or "row_feed_button_like" in post_xml_lower:
|
||||
# Specific check for opening a post (from explore/profile grid)
|
||||
if "view a post" in intent_lower or "first image" in intent_lower or "grid item" in intent_lower:
|
||||
if (
|
||||
"row_feed_photo_imageview" in post_xml_lower
|
||||
or "row_feed_button_like" in post_xml_lower
|
||||
or "clips_viewer_view_pager" in post_xml_lower
|
||||
):
|
||||
return True
|
||||
if (
|
||||
"explore_action_bar" in post_xml_lower
|
||||
and "row_feed_button_like" not in post_xml_lower
|
||||
and "clips_viewer" not in post_xml_lower
|
||||
):
|
||||
logger.warning(f"⚠️ [ActionMemory] Still on grid after trying to '{intent}'. Verification FAIL.")
|
||||
return False # Still on grid, definitely failed
|
||||
|
||||
# Specific check for opening a profile
|
||||
if "profile" in intent_lower or "author" in intent_lower or "username" in intent_lower:
|
||||
if "profile_header_container" in post_xml_lower:
|
||||
logger.info("✅ [ActionMemory] Structural check confirmed profile navigation success.")
|
||||
return True
|
||||
else:
|
||||
logger.warning(
|
||||
f"⚠️ [ActionMemory] Profile header NOT found after trying to '{intent}'. Verification FAIL."
|
||||
)
|
||||
return False
|
||||
|
||||
# Specific check for navigating to Home Feed
|
||||
if "home feed" in intent_lower or "home tab" in intent_lower:
|
||||
if "main_feed_action_bar" in post_xml_lower:
|
||||
logger.info("✅ [ActionMemory] Structural check confirmed Home Feed navigation success.")
|
||||
return True
|
||||
|
||||
# Specific check for navigating to Explore Feed
|
||||
if "explore feed" in intent_lower or "explore tab" in intent_lower or "search" in intent_lower:
|
||||
if "explore_action_bar" in post_xml_lower or "action_bar_search_edit_text" in post_xml_lower:
|
||||
logger.info("✅ [ActionMemory] Structural check confirmed Explore Feed navigation success.")
|
||||
return True
|
||||
if "explore_action_bar" in post_xml_lower and "row_feed_button_like" not in post_xml_lower:
|
||||
return None # Still on grid, inconclusive
|
||||
|
||||
state_toggles = ["like", "save", "follow", "heart"]
|
||||
is_toggle = any(t in intent_lower for t in state_toggles)
|
||||
|
||||
# ── State-Specific Structural Verification ──
|
||||
# If it was a follow, the resulting XML MUST contain "Following", "Requested", "Abonniert" or "Angefragt"
|
||||
if "follow" in intent_lower:
|
||||
FOLLOW_SUCCESS_MARKERS = ["following", "requested", "abonniert", "angefragt", "gefolgt"]
|
||||
if any(m in post_xml_lower for m in FOLLOW_SUCCESS_MARKERS):
|
||||
logger.info("✅ [ActionMemory] Structural check confirmed follow success.")
|
||||
return True
|
||||
else:
|
||||
logger.warning("⚠️ [ActionMemory] Follow success markers NOT found in post-click XML.")
|
||||
# We don't return False immediately because it might take a second to update
|
||||
# ── VLM Verification Fallback ──
|
||||
|
||||
# If we are highly confident (e.g. pulled from Qdrant memory), bypass heavy VLM
|
||||
if device and confidence < 0.95:
|
||||
@@ -172,16 +236,27 @@ class ActionMemory:
|
||||
|
||||
try:
|
||||
screenshot = device.get_screenshot_b64()
|
||||
if not screenshot:
|
||||
raise ValueError("No screenshot available from device")
|
||||
response = evaluator._query_vlm(prompt, screenshot)
|
||||
|
||||
if response and "yes" in response.lower() and "no" not in response.lower():
|
||||
decision = _parse_yes_no(response) if response else None
|
||||
|
||||
if decision is True:
|
||||
logger.debug(f"🧠 [ActionMemory] VLM visually confirmed success for '{intent}'.")
|
||||
return True
|
||||
else:
|
||||
elif decision is False:
|
||||
logger.warning(
|
||||
f"⚠️ [ActionMemory] VLM visual verification FAILED for '{intent}'. VLM replied: '{response}'"
|
||||
)
|
||||
return False
|
||||
else:
|
||||
# VLM returned ambiguous response (JSON, mixed signals, etc.)
|
||||
# Don't treat as hard failure — fall through to structural delta verification
|
||||
logger.info(
|
||||
f"🧠 [ActionMemory] VLM response for '{intent}' was not YES/NO "
|
||||
f"(got: '{response[:80]}...'). Falling through to structural verification."
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to query VLM for visual verification: {e}")
|
||||
# Fallthrough to structural delta if VLM crashes
|
||||
@@ -196,7 +271,9 @@ class ActionMemory:
|
||||
return False
|
||||
|
||||
# Fallback to structural delta
|
||||
logger.info(f"DEBUG: len(pre_click_xml)={len(pre_click_xml)} len(post_click_xml)={len(post_click_xml)}")
|
||||
diff = abs(len(pre_click_xml) - len(post_click_xml))
|
||||
logger.info(f"DEBUG: diff={diff}")
|
||||
|
||||
if is_toggle:
|
||||
if diff > 1000:
|
||||
@@ -207,15 +284,62 @@ class ActionMemory:
|
||||
if diff > 0:
|
||||
logger.debug(f"🧠 [ActionMemory] Structural delta detected for toggle '{intent}'. Verification PASS.")
|
||||
return True
|
||||
else:
|
||||
logger.warning(
|
||||
f"⚠️ [ActionMemory] Zero structural shift (diff={diff}) for state-toggle '{intent}'. Verification FAIL."
|
||||
)
|
||||
return False
|
||||
# If the intent is an abstract goal (like "find customers"), diff > 50 is NOT enough.
|
||||
# We must force visual VLM confirmation because clicking the wrong thing (like "Create highlight")
|
||||
# also produces a large diff but achieves the wrong goal.
|
||||
if diff > 50:
|
||||
logger.debug(
|
||||
f"🧠 [ActionMemory] Structural change detected for navigation '{intent}'. Verification PASS."
|
||||
)
|
||||
return True
|
||||
# Is it a standard structural transition?
|
||||
from GramAddict.core.screen_topology import ScreenTopology
|
||||
|
||||
logger.warning(f"⚠️ [ActionMemory] No structural change detected for '{intent}'. Verification FAIL.")
|
||||
return False
|
||||
# We don't have screen type here, so we just check if it's in the HD Map keys
|
||||
logger.info(f"DEBUG: intent is '{intent}'")
|
||||
logger.info(
|
||||
f"DEBUG: TRANSITIONS keys are: {[list(t.keys()) for t in ScreenTopology.TRANSITIONS.values()]}"
|
||||
)
|
||||
is_standard = any(intent in transitions for transitions in ScreenTopology.TRANSITIONS.values())
|
||||
logger.info(f"DEBUG: is_standard={is_standard}")
|
||||
|
||||
if is_standard:
|
||||
logger.debug(
|
||||
f"🧠 [ActionMemory] Structural change detected for known navigation '{intent}'. Verification PASS."
|
||||
)
|
||||
return True
|
||||
else:
|
||||
logger.info(
|
||||
f"👁️ [ActionMemory] Abstract intent '{intent}' caused UI change. Forcing VLM visual verification..."
|
||||
)
|
||||
# For abstract intents, we must visually verify if it actually helped!
|
||||
# If device is available, we use VLM. If not, we fail safe.
|
||||
if device:
|
||||
from GramAddict.core.perception.semantic_evaluator import SemanticEvaluator
|
||||
|
||||
evaluator = SemanticEvaluator()
|
||||
prompt = f"The user just attempted to perform the action: '{intent}'. Does the current screen match the expected outcome? Answer ONLY with the word YES or NO."
|
||||
try:
|
||||
response = evaluator._query_vlm(prompt, device.get_screenshot_b64())
|
||||
decision = _parse_yes_no(response) if response else None
|
||||
if decision is True:
|
||||
return True
|
||||
else:
|
||||
logger.warning(
|
||||
f"⚠️ [ActionMemory] VLM rejected success for abstract intent '{intent}'. Response: '{response}'"
|
||||
)
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f"VLM visual verification failed: {e}")
|
||||
|
||||
logger.warning(f"⚠️ [ActionMemory] Cannot visually verify abstract intent '{intent}'. Failing safe.")
|
||||
return False
|
||||
|
||||
# If diff <= 50 for non-toggle
|
||||
logger.warning(
|
||||
f"⚠️ [ActionMemory] Insufficient structural change (diff={diff}) for non-toggle '{intent}'. Verification FAIL."
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
def _intent_matches_node(intent: str, semantic_string: str) -> bool:
|
||||
|
||||
@@ -46,15 +46,15 @@ def has_carousel_in_view(xml_dump: str) -> bool:
|
||||
return any(ind in xml_dump for ind in CAROUSEL_INDICATORS)
|
||||
|
||||
|
||||
def extract_post_content(context_xml: str) -> dict:
|
||||
def extract_post_content(context_xml: str, device=None) -> dict:
|
||||
"""
|
||||
Extracts meaningful content data from the current feed post's XML.
|
||||
This is the BOT'S EYES — what it actually "sees" about each post.
|
||||
|
||||
Returns:
|
||||
{'username': str, 'description': str, 'caption': str}
|
||||
{'username': str, 'description': str, 'caption': str, 'username_missing': bool}
|
||||
"""
|
||||
result = {"username": "", "description": "", "caption": ""}
|
||||
result = {"username": "", "description": "", "caption": "", "username_missing": False}
|
||||
|
||||
try:
|
||||
from GramAddict.core.telepathic_engine import TelepathicEngine
|
||||
@@ -62,16 +62,87 @@ def extract_post_content(context_xml: str) -> dict:
|
||||
telepath = TelepathicEngine.get_instance()
|
||||
|
||||
# 1. Learn/extract post author dynamically
|
||||
author_node = telepath.find_best_node(context_xml, "post author username header", min_confidence=0.75)
|
||||
# 🛡️ Structural Fast-Path: Prioritize deterministic IDs over AI guesses
|
||||
# Try structural ID fast-path first (100% deterministic)
|
||||
author_node = None
|
||||
try:
|
||||
root = ET.fromstring(context_xml)
|
||||
for node in root.iter("node"):
|
||||
res_id = node.attrib.get("resource-id", "")
|
||||
if "row_feed_photo_profile_name" in res_id or "clips_author_username" in res_id or "profile_header_name" in res_id:
|
||||
author_node = {"original_attribs": node.attrib}
|
||||
logger.debug(f"Identified author_node via structural ID: {res_id}")
|
||||
break
|
||||
except Exception as e:
|
||||
logger.debug(f"XML parse error in author structural fast-path: {e}")
|
||||
|
||||
# 🛡️ Anti-Hallucination Guard: The author header is always near the top. Ignore names in the comment section.
|
||||
if author_node and author_node.get("y", 0) < 1000 and author_node.get("original_attribs", {}).get("text"):
|
||||
result["username"] = author_node["original_attribs"]["text"].strip()
|
||||
# Fallback to Telepathic Engine if structural ID is missing
|
||||
if not author_node:
|
||||
author_node = telepath.find_best_node(
|
||||
context_xml, "post author username text (exclude bottom tabs)", min_confidence=0.75, device=device
|
||||
)
|
||||
logger.debug(f"Telepathic fallback for author_node: {author_node}")
|
||||
|
||||
# 🛡️ Anti-Hallucination Guard: Ensure we actually found text.
|
||||
if author_node:
|
||||
attribs = author_node.get("original_attribs", {})
|
||||
text = attribs.get("text", "").strip()
|
||||
desc = attribs.get("content_desc", "").strip()
|
||||
|
||||
if text:
|
||||
result["username"] = text
|
||||
elif desc:
|
||||
result["username"] = desc
|
||||
else:
|
||||
# If the VLM selected a container (like clips_author_info_component),
|
||||
# extract text from its children.
|
||||
logger.debug("Author node lacks text/desc. Searching children for username...")
|
||||
bounds = attribs.get("bounds")
|
||||
if bounds:
|
||||
try:
|
||||
# Re-parse to find children within bounds
|
||||
import re
|
||||
match = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", bounds)
|
||||
if match:
|
||||
l, t, r, b = map(int, match.groups())
|
||||
|
||||
# Fallback: scan all nodes in XML and see if they are inside these bounds
|
||||
possible_texts = []
|
||||
possible_descs = []
|
||||
for n in ET.fromstring(context_xml).iter("node"):
|
||||
child_bounds = n.attrib.get("bounds")
|
||||
child_text = n.attrib.get("text", "").strip()
|
||||
child_desc = n.attrib.get("content-desc", "").strip()
|
||||
|
||||
if child_bounds and (child_text or child_desc):
|
||||
cm = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", child_bounds)
|
||||
if cm:
|
||||
cl, ct, cr, cb = map(int, cm.groups())
|
||||
# Check if child is strictly inside the container
|
||||
if cl >= l and ct >= t and cr <= r and cb <= b:
|
||||
if child_text:
|
||||
possible_texts.append(child_text)
|
||||
if child_desc and "Profile picture" not in child_desc:
|
||||
possible_descs.append(child_desc)
|
||||
|
||||
if possible_texts:
|
||||
result["username"] = possible_texts[0]
|
||||
logger.debug(f"Extracted username '{result['username']}' from child node text.")
|
||||
elif possible_descs:
|
||||
result["username"] = possible_descs[0]
|
||||
logger.debug(f"Extracted username '{result['username']}' from child node desc.")
|
||||
except Exception as e:
|
||||
logger.debug(f"Failed to extract username from children: {e}")
|
||||
|
||||
# 2. Learn/extract post media description dynamically
|
||||
media_node = telepath.find_best_node(context_xml, "post media content", min_confidence=0.35)
|
||||
if media_node and media_node.get("original_attribs", {}).get("desc"):
|
||||
result["description"] = media_node["original_attribs"]["desc"].strip()
|
||||
media_node = telepath.find_best_node(
|
||||
context_xml,
|
||||
"post media content (the actual image or video, exclude bottom tabs)",
|
||||
min_confidence=0.35,
|
||||
device=device,
|
||||
)
|
||||
if media_node and media_node.get("original_attribs", {}).get("content_desc"):
|
||||
result["description"] = media_node["original_attribs"]["content_desc"].strip()
|
||||
|
||||
# 3. Visible caption text (heuristic fallback if node isn't explicitly found)
|
||||
# Search all nodes for text that contains the username to find the caption body
|
||||
@@ -85,6 +156,11 @@ def extract_post_content(context_xml: str) -> dict:
|
||||
except Exception as e:
|
||||
logger.warning(f"Error extracting post content autonomously: {e}")
|
||||
|
||||
# REGRESSION FIX 2026-05-01: Flag unreliable data when username is empty
|
||||
if not result["username"]:
|
||||
result["username_missing"] = True
|
||||
logger.warning("⚠️ [PostDataExtraction] Username is empty — data may be unreliable.")
|
||||
|
||||
return result
|
||||
|
||||
|
||||
|
||||
@@ -8,15 +8,17 @@ from GramAddict.core.perception.spatial_parser import SpatialNode
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Navigation tab intent → resource_id keyword mapping
|
||||
# These are STRUCTURAL guards (bottom 15% zone), not string-matching heuristics.
|
||||
_NAV_TAB_MAP = {
|
||||
"tap home tab": "feed_tab",
|
||||
"tap explore tab": "search_tab",
|
||||
"tap reels tab": "clips_tab",
|
||||
"tap profile tab": "profile_tab",
|
||||
"tap messages tab": "direct_tab",
|
||||
}
|
||||
|
||||
def _humanize_desc(desc: str) -> str:
|
||||
"""
|
||||
Inserts a space between numbers and letters to fix Instagram's concatenated content-desc.
|
||||
Example: "991following" -> "991 following", "140Kfollowers" -> "140K followers"
|
||||
"""
|
||||
if not desc:
|
||||
return ""
|
||||
import re
|
||||
|
||||
return re.sub(r"(\d[KMBkmb]?)([a-z])", r"\1 \2", desc)
|
||||
|
||||
|
||||
class IntentResolver:
|
||||
@@ -34,47 +36,396 @@ class IntentResolver:
|
||||
3. Fallback → text-based VLM (when no device/screenshot available)
|
||||
"""
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Structural Guards
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def filter_navigation_conflicts(
|
||||
self, candidates: List[SpatialNode], intent_description: str, screen_height: int = 2400
|
||||
) -> List[SpatialNode]:
|
||||
"""
|
||||
Prevents VLM from confusing navigation-bar buttons (Back, Close)
|
||||
with bottom tab-bar buttons (Home, Profile, Search).
|
||||
|
||||
Production bug 2026-04-30: VLM picked action_bar_button_back
|
||||
for "tap profile tab" → account switch failed.
|
||||
|
||||
Production bug 2026-05-01: VLM picked profile_tab (desc='Profile')
|
||||
for "post author username text" → navigated to own profile instead.
|
||||
|
||||
Rules:
|
||||
- For tab intents: exclude nodes with "back" in resource_id or
|
||||
content_desc == "Back"
|
||||
- For back/close intents: no filtering (Back is the correct target)
|
||||
- For author/username intents: exclude bottom navigation tabs
|
||||
"""
|
||||
intent_lower = intent_description.lower()
|
||||
|
||||
# Only apply for REAL tab navigation intents.
|
||||
# REGRESSION FIX 2026-05-02: "tab" as a substring was too broad.
|
||||
# Intent "post author username text (exclude bottom tabs)" matched
|
||||
# because it contained "tab" → Tab Height Guard nuked the author node.
|
||||
# Now we require specific tab navigation patterns:
|
||||
# - "tap profile tab", "tap home tab", "explore tab"
|
||||
# - NOT "exclude bottom tabs", "tabbar", random mentions
|
||||
import re
|
||||
|
||||
_TAB_PATTERN = re.compile(
|
||||
r"\btap\s+\w+\s+tab\b" # "tap profile tab", "tap home tab"
|
||||
r"|\b\w+\s+tab\b" # "profile tab", "explore tab"
|
||||
r"|^tab\b", # "tab" at start of intent
|
||||
re.IGNORECASE,
|
||||
)
|
||||
filtered = []
|
||||
is_tab_intent = bool(_TAB_PATTERN.search(intent_lower)) and "back" not in intent_lower
|
||||
is_create_intent = "create" in intent_lower or "camera" in intent_lower or "story" in intent_lower
|
||||
# REGRESSION FIX 2026-05-01: Author/username intents must never pick nav tabs
|
||||
is_author_intent = any(kw in intent_lower for kw in ["author", "username", "post media"])
|
||||
|
||||
# Known bottom navigation tab resource_id suffixes
|
||||
NAV_TAB_SUFFIXES = ("_tab", "tab_icon", "navigation_bar")
|
||||
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
desc = (node.content_desc or "").lower()
|
||||
|
||||
is_back = "back" in rid or desc == "back"
|
||||
is_close = "close" in rid or desc == "close"
|
||||
is_create = "camera" in rid or "create" in rid or desc == "camera" or desc == "create" or "creation" in rid
|
||||
is_nav_tab = any(rid.endswith(s) for s in NAV_TAB_SUFFIXES)
|
||||
|
||||
if is_tab_intent and (is_back or is_close):
|
||||
logger.debug(
|
||||
f"🛡️ [Nav Conflict Guard] Excluded '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for tab intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
# NEW REGRESSION FIX 2026-05-01: Tab intents must never pick top-screen elements or headers
|
||||
# UPDATE: Actually, tabs are always at the very bottom. Filter out anything above 85% of screen height.
|
||||
if is_tab_intent:
|
||||
is_not_at_bottom = node.center_y < (screen_height * 0.85)
|
||||
if is_not_at_bottom:
|
||||
logger.debug(
|
||||
f"🛡️ [Tab Height Guard] Excluded non-bottom element '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}', y={node.center_y}) for tab intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
# Tab intents should NEVER be content items
|
||||
if any(kw in desc for kw in ["reel by", "photo by", "photos by", "row ", "column "]):
|
||||
logger.debug(
|
||||
f"🛡️ [Content Tab Guard] Excluded content item '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for tab intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
if is_author_intent and is_nav_tab:
|
||||
logger.debug(
|
||||
f"🛡️ [Author Tab Guard] Excluded nav tab '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for author intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
if not is_create_intent and is_create:
|
||||
logger.debug(
|
||||
f"🛡️ [Creation Conflict Guard] Excluded '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
# NEW REGRESSION FIX: Exclude interaction buttons (comment, like, share) when looking for author or media
|
||||
# This prevents the weak VLM from hallucinating bounding box numbers that point to "Comment".
|
||||
interaction_suffixes = ["comment", "like", "share", "send", "save", "button_icon"]
|
||||
is_interaction = any(s in rid for s in interaction_suffixes) or any(s in desc for s in interaction_suffixes)
|
||||
node_text_lower = (node.text or "").lower()
|
||||
is_follow = "follow" in rid or "follow" in node_text_lower or "follow" in desc
|
||||
is_media_intent = "media content" in intent_lower or "image" in intent_lower or "video" in intent_lower
|
||||
|
||||
if is_author_intent and (is_interaction or is_follow):
|
||||
logger.debug(
|
||||
f"🛡️ [Author Interaction Guard] Excluded interaction/follow button '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}', text='{node.text}') for author intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
if is_media_intent and (is_interaction or is_follow):
|
||||
logger.debug(
|
||||
f"🛡️ [Media Interaction Guard] Excluded interaction/follow button '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for media intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
# NEW REGRESSION FIX: Exclude stories and reels tray when looking for a POST author
|
||||
# This prevents the VLM from selecting the user's own story at the top of the feed
|
||||
is_post_author_intent = is_author_intent and "post" in intent_lower
|
||||
node_text_lower = (node.text or "").lower()
|
||||
is_story_or_reel = (
|
||||
"story" in rid
|
||||
or "story" in desc
|
||||
or "story" in node_text_lower
|
||||
or "reel" in rid
|
||||
or "reel" in desc
|
||||
or "reel" in node_text_lower
|
||||
)
|
||||
|
||||
if is_post_author_intent and is_story_or_reel:
|
||||
logger.debug(
|
||||
f"🛡️ [Post Author Story Guard] Excluded story/reel '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for post author intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
# NEW REGRESSION FIX: Exclude action bar titles (like 'For you') when looking for an author
|
||||
if is_author_intent and "action_bar_title" in rid:
|
||||
logger.debug(
|
||||
f"🛡️ [Author Action Bar Guard] Excluded action bar title '{node.resource_id}' "
|
||||
f"(desc='{node.content_desc}') for author intent '{intent_description}'"
|
||||
)
|
||||
continue
|
||||
|
||||
filtered.append(node)
|
||||
|
||||
return filtered
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Public API
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def resolve(
|
||||
self, intent_description: str, candidates: List[SpatialNode], screen_height: int = 2400, device=None
|
||||
self, intent_description: str, candidates: List[SpatialNode], device=None, screen_height: int = 2400
|
||||
) -> Optional[SpatialNode]:
|
||||
if not candidates:
|
||||
return None
|
||||
|
||||
intent_lower = intent_description.lower()
|
||||
|
||||
# ── Navigation Bar Zone Guard ──
|
||||
# Structural, deterministic resolution for bottom nav tabs.
|
||||
tab_keyword = _NAV_TAB_MAP.get(intent_lower)
|
||||
if tab_keyword:
|
||||
nav_zone_y = int(screen_height * 0.85)
|
||||
nav_candidates = [
|
||||
n for n in candidates if n.y1 >= nav_zone_y and tab_keyword in (n.resource_id or "").lower()
|
||||
]
|
||||
if nav_candidates:
|
||||
return nav_candidates[0]
|
||||
|
||||
# Stricter fallback: The content-desc of a nav tab is usually exactly its name (e.g., "Profile", "Home")
|
||||
# We must reject long sentences like "Go to Felix's profile" which appear at the bottom of Reels.
|
||||
tab_label = intent_lower.replace("tap ", "").replace(" tab", "").strip()
|
||||
nav_candidates = [
|
||||
n for n in candidates if n.y1 >= nav_zone_y and (n.content_desc or "").lower() == tab_label
|
||||
]
|
||||
if nav_candidates:
|
||||
return nav_candidates[0]
|
||||
return None
|
||||
|
||||
# Block abstract goals from leaking into node clicks
|
||||
abstract_goals = ["open profile", "open explore", "open following", "learn own profile"]
|
||||
if intent_lower in abstract_goals:
|
||||
return None
|
||||
|
||||
# --- Strict VLM Hallucination Guard ---
|
||||
# For known structural targets that the VLM frequently hallucinates when they are missing,
|
||||
# we enforce a strict failure if they weren't caught by the structural fast paths.
|
||||
# --- Strict Structural Fast-Paths ---
|
||||
# Bypass VLM for deterministically identifiable UI components
|
||||
if "message text box" in intent_lower or "message input" in intent_lower or "type message" in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
text = (node.text or "").lower()
|
||||
if "composer_edittext" in rid or "message…" in text or "message..." in text:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found message input field: {rid}")
|
||||
return node
|
||||
|
||||
if "last received message text" in intent_lower or "received message" in intent_lower:
|
||||
# Gather all message text views
|
||||
msg_nodes = [n for n in candidates if "direct_text_message_text_view" in (n.resource_id or "").lower()]
|
||||
if msg_nodes:
|
||||
# The last one in the XML is typically the most recent message at the bottom of the screen
|
||||
latest_msg = msg_nodes[-1]
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found last received message text: '{latest_msg.text}'")
|
||||
return latest_msg
|
||||
|
||||
if "send message button" in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
desc = (node.content_desc or "").lower()
|
||||
text = (node.text or "").lower()
|
||||
if "send" in rid or "composer_button" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found send button: {rid or desc or text}")
|
||||
return node
|
||||
|
||||
if "post author username" in intent_lower or "tap post username" in intent_lower:
|
||||
for node in candidates:
|
||||
if "row_feed_photo_profile_imageview" in (node.resource_id or "").lower():
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found post author avatar image: {node.content_desc}")
|
||||
return node
|
||||
for node in candidates:
|
||||
if "row_feed_photo_profile_name" in (node.resource_id or "").lower():
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found post author username text: {node.text}")
|
||||
return node
|
||||
|
||||
if "feed post content" in intent_lower or "post media content" in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if "row_feed_photo_imageview" in rid or "zoomable_view_container" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found feed post content: {rid}")
|
||||
return node
|
||||
|
||||
if "comment" in intent_lower and "button" in intent_lower:
|
||||
# First try the View all comments button
|
||||
for node in candidates:
|
||||
if (
|
||||
"view all comments" in (node.text or "").lower()
|
||||
or "view all comments" in (node.content_desc or "").lower()
|
||||
):
|
||||
logger.info(
|
||||
f"🎯 [Structural Fast-Path] Found comment button text: {node.text or node.content_desc}"
|
||||
)
|
||||
return node
|
||||
# Then try the icon itself if somehow clickable
|
||||
for node in candidates:
|
||||
if (
|
||||
"row_feed_button_comment" in (node.resource_id or "").lower()
|
||||
or "row_feed_textview_comments" in (node.resource_id or "").lower()
|
||||
):
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found comment button: {node.resource_id}")
|
||||
return node
|
||||
|
||||
if "like" in intent_lower and ("button" in intent_lower or "post" in intent_lower):
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if "row_feed_button_like" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found like button: {rid}")
|
||||
return node
|
||||
|
||||
if ("send" in intent_lower or "share" in intent_lower) and "post" in intent_lower and "button" in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
desc = (node.content_desc or "").lower()
|
||||
if "row_feed_button_share" in rid or "send post" in desc:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found send/share post button: {rid or desc}")
|
||||
return node
|
||||
|
||||
if "add to story" in intent_lower:
|
||||
# We skip structural fast-path for 'add to story' since it relies heavily on language/text strings
|
||||
# and let the VLM figure it out or rely on purely visual indicators.
|
||||
pass
|
||||
|
||||
if "share" in intent_lower and ("button" in intent_lower or "post" in intent_lower):
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if "row_feed_button_share" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found share button: {rid}")
|
||||
return node
|
||||
|
||||
if "save" in intent_lower and ("button" in intent_lower or "post" in intent_lower):
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if "row_feed_button_save" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found save button: {rid}")
|
||||
return node
|
||||
|
||||
if "follow" in intent_lower and "button" in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if "profile_header_follow_button" in rid or "inline_follow_button" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found follow/following button: {rid}")
|
||||
return node
|
||||
|
||||
if "first post" in intent_lower or "first item" in intent_lower or "first search result" in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if "grid_card_layout_container" in rid or "image_button" in rid or "row_search_user" in rid:
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found first post/item: {rid}")
|
||||
return node
|
||||
|
||||
if "story ring" in intent_lower or "story tray" in intent_lower:
|
||||
story_nodes = []
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
desc = (node.content_desc or "").lower()
|
||||
text = (node.text or "").lower()
|
||||
# Instagram story tray avatars usually have this resource id and 'story' in the content description
|
||||
if ("avatar_image_view" in rid or "row_profile_header_imageview" in rid) and "story" in desc:
|
||||
# Ignore the user's explicit "Add to story" ring
|
||||
if "add to story" not in desc and "your story" not in text:
|
||||
story_nodes.append(node)
|
||||
|
||||
if story_nodes:
|
||||
# Sort horizontally (left-to-right)
|
||||
story_nodes.sort(key=lambda n: n.x1)
|
||||
|
||||
# Check if this is the home feed story tray (avatar_image_view without 'highlight' in desc)
|
||||
is_highlight = any("highlight" in (n.content_desc or "").lower() for n in story_nodes)
|
||||
if "avatar_image_view" in (story_nodes[0].resource_id or "").lower() and not is_highlight:
|
||||
if len(story_nodes) > 1:
|
||||
logger.info(
|
||||
f"🎯 [Structural Fast-Path] Found {len(story_nodes)} story rings. Skipping own profile. Picking second: '{story_nodes[1].content_desc}'"
|
||||
)
|
||||
return story_nodes[1]
|
||||
else:
|
||||
logger.warning(
|
||||
"🎯 [Structural Fast-Path] Only 1 story ring found on feed (likely own profile). Skipping to avoid modal trap."
|
||||
)
|
||||
return None
|
||||
else:
|
||||
# Profile header or other single-story views
|
||||
logger.info(
|
||||
f"🎯 [Structural Fast-Path] Found story ring avatar: {story_nodes[0].resource_id} (desc: '{story_nodes[0].content_desc}')"
|
||||
)
|
||||
return story_nodes[0]
|
||||
|
||||
# --- Navigation Tab Fast-Paths ---
|
||||
# Deterministically identify bottom navigation tabs to prevent VLM confusion
|
||||
tab_map = {
|
||||
"home tab": "feed_tab",
|
||||
"feed tab": "feed_tab",
|
||||
"reels tab": "clips_tab",
|
||||
"clips tab": "clips_tab",
|
||||
"explore tab": "search_tab",
|
||||
"search tab": "search_tab",
|
||||
"profile tab": "profile_tab",
|
||||
"message tab": "direct_tab",
|
||||
"direct tab": "direct_tab",
|
||||
}
|
||||
for intent_key, resource_suffix in tab_map.items():
|
||||
if intent_key in intent_lower:
|
||||
for node in candidates:
|
||||
rid = (node.resource_id or "").lower()
|
||||
if rid.endswith(f":id/{resource_suffix}"):
|
||||
logger.info(f"🎯 [Structural Fast-Path] Found {intent_key}: {rid}")
|
||||
return node
|
||||
|
||||
# --- Semantic Match Guard ---
|
||||
# If the intent explicitly quotes a target (e.g., "tap 'New Message'"),
|
||||
# we strictly filter candidates to those whose text or content_desc contains the quote.
|
||||
import re
|
||||
|
||||
quotes = re.findall(r"['\"](.*?)['\"]", intent_description)
|
||||
if quotes:
|
||||
target_text = quotes[0].lower()
|
||||
|
||||
# Only use the exact target string (no manual localized translation dictionaries!)
|
||||
localized_targets = [target_text]
|
||||
|
||||
semantic_candidates = []
|
||||
for node in candidates:
|
||||
n_text = (node.text or "").lower()
|
||||
n_desc = (node.content_desc or "").lower()
|
||||
|
||||
# Check if any of the localized targets match
|
||||
for loc_target in localized_targets:
|
||||
pattern = r"\b" + re.escape(loc_target) + r"\b"
|
||||
if re.search(pattern, n_text) or re.search(pattern, n_desc):
|
||||
semantic_candidates.append(node)
|
||||
break # Found a match, no need to check other localized targets
|
||||
|
||||
if semantic_candidates:
|
||||
if len(semantic_candidates) == 1:
|
||||
logger.info(f"🎯 [Semantic Guard] Exact match found for '{target_text}', skipping VLM.")
|
||||
return semantic_candidates[0]
|
||||
else:
|
||||
logger.info(
|
||||
f"🎯 [Semantic Guard] {len(semantic_candidates)} matches found for '{target_text}'. Reducing candidates for VLM."
|
||||
)
|
||||
candidates = semantic_candidates
|
||||
else:
|
||||
logger.warning(
|
||||
f"⚠️ [Semantic Guard] No candidates found containing '{target_text}'. Returning None to prevent hallucination."
|
||||
)
|
||||
return None
|
||||
|
||||
# ── PRIMARY PATH: Visual Discovery ──
|
||||
# If we have a device, the VLM SEES the screen and decides.
|
||||
if device is not None and (
|
||||
hasattr(device, "screenshot") or hasattr(getattr(device, "deviceV2", None), "screenshot")
|
||||
):
|
||||
print(f"DEBUG_INTENT: Entering Visual Discovery for '{intent_description}'")
|
||||
logger.info("📸 Device screenshot capability detected. Enforcing visual discovery.")
|
||||
return self._visual_discovery(intent_description, candidates, device, screen_height=screen_height)
|
||||
|
||||
print(f"DEBUG_INTENT: Falling back to Text-based VLM for '{intent_description}'")
|
||||
# --- Strict VLM Hallucination Guard (Text-only Fallback) ---
|
||||
# For known structural targets that the text-based VLM frequently hallucinates when they are missing,
|
||||
# we enforce a strict failure.
|
||||
if "following list" in intent_lower or "followers list" in intent_lower or "tap message button" in intent_lower:
|
||||
logger.warning(
|
||||
f"🛡️ [Hallucination Guard] Intent '{intent_description}' is a strict structural target. "
|
||||
@@ -82,17 +433,9 @@ class IntentResolver:
|
||||
)
|
||||
return None
|
||||
|
||||
# ── PRIMARY PATH: Visual Discovery ──
|
||||
# If we have a device, the VLM SEES the screen and decides.
|
||||
if device:
|
||||
result = self._visual_discovery(intent_description, candidates, device)
|
||||
if result:
|
||||
return result
|
||||
logger.warning(f"👁️ [Visual Discovery] No match found for '{intent_description}', trying text fallback.")
|
||||
|
||||
# ── FALLBACK: Text-based VLM resolution ──
|
||||
# Only used when device is unavailable (e.g., unit tests without screenshots).
|
||||
return self._text_based_resolve(intent_description, candidates, device)
|
||||
return self._text_based_resolve(intent_description, candidates, device, screen_height=screen_height)
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Visual Discovery (Set-of-Mark Prompting)
|
||||
@@ -112,15 +455,8 @@ class IntentResolver:
|
||||
|
||||
img = device.deviceV2.screenshot()
|
||||
|
||||
# Stage 1: Basic area filter + exclude system UI and notifications
|
||||
pre_filtered = [
|
||||
n
|
||||
for n in candidates
|
||||
if 200 < n.area < 400000
|
||||
and "com.android.systemui" not in (n.resource_id or "")
|
||||
and "notification:" not in (n.content_desc or "").lower()
|
||||
and "per cent" not in (n.content_desc or "").lower()
|
||||
]
|
||||
# Stage 1: Basic area filter + exclude system UI and notifications (ALREADY HANDLED in _visual_discovery)
|
||||
pre_filtered = candidates
|
||||
|
||||
# Stage 2: Spatial deduplication
|
||||
# A node could completely contain another.
|
||||
@@ -227,7 +563,7 @@ class IntentResolver:
|
||||
return annotated_b64, box_map
|
||||
|
||||
def _visual_discovery(
|
||||
self, intent_description: str, candidates: List[SpatialNode], device
|
||||
self, intent_description: str, candidates: List[SpatialNode], device, screen_height: int = 2400
|
||||
) -> Optional[SpatialNode]:
|
||||
"""
|
||||
Vision-first intent resolution via Set-of-Mark (SoM) prompting.
|
||||
@@ -241,6 +577,21 @@ class IntentResolver:
|
||||
from GramAddict.core.config import Config
|
||||
from GramAddict.core.llm_provider import query_telepathic_llm
|
||||
|
||||
# Pre-filter candidates by area and system UI before any semantic matching
|
||||
candidates = [
|
||||
n
|
||||
for n in candidates
|
||||
if 200 < n.area < 400000
|
||||
and "com.android.systemui" not in (n.resource_id or "")
|
||||
and "notification:" not in (n.content_desc or "").lower()
|
||||
and "per cent" not in (n.content_desc or "").lower()
|
||||
]
|
||||
|
||||
# --- Navigation Conflict Guard ---
|
||||
# Prevents VLM from confusing Back buttons with tab buttons
|
||||
# Production bug 2026-04-30: VLM picked Back for "tap profile tab"
|
||||
candidates = self.filter_navigation_conflicts(candidates, intent_description, screen_height=screen_height)
|
||||
|
||||
# --- Strict Button Guard ---
|
||||
# If the intent specifically asks for a "button", "icon", or "tab",
|
||||
# filter out candidates that contain long text (e.g. captions, comments)
|
||||
@@ -256,40 +607,46 @@ class IntentResolver:
|
||||
logger.debug(f"🛡️ [Strict Button Guard] Filtered out node with long text: '{node.text[:20]}...'")
|
||||
candidates = filtered_candidates
|
||||
|
||||
# --- Semantic Match Guard ---
|
||||
# If the intent explicitly quotes a target (e.g., "tap 'New Message'"),
|
||||
# we strictly filter candidates to those whose text or content_desc contains the quote.
|
||||
import re
|
||||
|
||||
quotes = re.findall(r"['\"](.*?)['\"]", intent_description)
|
||||
if quotes:
|
||||
target_text = quotes[0].lower()
|
||||
pattern = r"\b" + re.escape(target_text) + r"\b"
|
||||
semantic_candidates = []
|
||||
# --- Post/Grid Item Guard ---
|
||||
# VLMs frequently hallucinate 'Search' when asked to tap a post. We must pre-filter.
|
||||
if "first post" in intent_lower or "grid item" in intent_lower:
|
||||
grid_candidates = []
|
||||
for node in candidates:
|
||||
n_text = (node.text or "").lower()
|
||||
n_desc = (node.content_desc or "").lower()
|
||||
if re.search(pattern, n_text) or re.search(pattern, n_desc):
|
||||
semantic_candidates.append(node)
|
||||
desc = (node.content_desc or "").lower()
|
||||
# Posts/grid items usually have 'row X, column Y', 'photos by', or 'reel by'
|
||||
if "row 1" in desc or "column" in desc or "photos by" in desc or "reel by" in desc:
|
||||
grid_candidates.append(node)
|
||||
|
||||
if semantic_candidates:
|
||||
if len(semantic_candidates) == 1:
|
||||
logger.info(f"🎯 [Semantic Guard] Exact match found for '{target_text}', skipping VLM.")
|
||||
return semantic_candidates[0]
|
||||
else:
|
||||
logger.info(
|
||||
f"🎯 [Semantic Guard] {len(semantic_candidates)} matches found for '{target_text}'. Reducing candidates for VLM."
|
||||
if grid_candidates:
|
||||
logger.info(f"🎯 [Grid Guard] Filtered to {len(grid_candidates)} actual grid candidates.")
|
||||
candidates = grid_candidates
|
||||
|
||||
# --- Author/Username Guard ---
|
||||
# Prevents VLM from picking the "Profile" nav tab when asked for "post author username".
|
||||
if "author" in intent_lower or "username" in intent_lower or "profile name" in intent_lower:
|
||||
filtered_candidates = []
|
||||
for node in candidates:
|
||||
res_id = (node.resource_id or "").lower()
|
||||
desc = (node.content_desc or "").lower()
|
||||
if (
|
||||
"tab" in res_id
|
||||
or "navigation" in res_id
|
||||
or "tabbar" in res_id
|
||||
or desc in ["home", "search", "reels", "profile"]
|
||||
):
|
||||
logger.debug(
|
||||
f"🛡️ [Author Guard] Filtered out navigation tab: '{node.content_desc}' ({node.resource_id})"
|
||||
)
|
||||
candidates = semantic_candidates
|
||||
else:
|
||||
logger.warning(
|
||||
f"⚠️ [Semantic Guard] No candidates found containing '{target_text}'. Returning None to prevent hallucination."
|
||||
)
|
||||
return None
|
||||
else:
|
||||
filtered_candidates.append(node)
|
||||
candidates = filtered_candidates
|
||||
|
||||
try:
|
||||
annotated_b64, box_map = self._annotate_screenshot_with_candidates(device, candidates)
|
||||
except Exception as e:
|
||||
import traceback
|
||||
|
||||
traceback.print_exc()
|
||||
logger.warning(f"⚠️ [Visual Discovery] Screenshot annotation failed: {e}")
|
||||
return None
|
||||
|
||||
@@ -308,15 +665,16 @@ class IntentResolver:
|
||||
node = box_map[idx]
|
||||
label_parts = []
|
||||
if node.content_desc:
|
||||
label_parts.append(f"desc='{node.content_desc[:50]}'")
|
||||
desc = _humanize_desc(node.content_desc)
|
||||
label_parts.append(f"desc='{desc[:50]}'")
|
||||
if node.text and node.text != node.content_desc:
|
||||
label_parts.append(f"text='{node.text[:50]}'")
|
||||
text = _humanize_desc(node.text)
|
||||
label_parts.append(f"text='{text[:50]}'")
|
||||
if not label_parts:
|
||||
label_parts.append("(no visible text)")
|
||||
box_legend_lines.append(f" [{idx}] {', '.join(label_parts)}")
|
||||
box_legend = "\n".join(box_legend_lines)
|
||||
print("BOX LEGEND:")
|
||||
print(box_legend)
|
||||
logger.debug(f"BOX LEGEND:\n{box_legend}")
|
||||
|
||||
prompt = (
|
||||
f"You are looking at a mobile app screenshot with numbered bounding boxes drawn around interactive UI elements.\n"
|
||||
@@ -330,7 +688,30 @@ class IntentResolver:
|
||||
f" - 'comment button' = SPEECH BUBBLE ICON, usually has desc='Comment'.\n"
|
||||
f"3. Do NOT select text, captions, or view counts if looking for an icon.\n"
|
||||
f"4. Ignore numbers inside the text itself. Do not confuse the text '19' with Box [19].\n"
|
||||
f"5. If the exact control is NOT visible, return null. Do NOT guess.\n\n"
|
||||
f"5. If the intent contains 'following', you MUST pick the box containing 'following'. Do NOT pick 'followers' or 'Follow'.\n"
|
||||
f"6. If the intent is to tap a 'post', 'first post', or 'grid item':\n"
|
||||
f" - Look for boxes with descriptions containing 'photos by', 'Reel by', or 'row 1, column 1'.\n"
|
||||
f" - Pick the FIRST matching box index (e.g. if [0] says '6 photos...', return 0, NOT 6).\n"
|
||||
f" - Do NOT pick navigation buttons like 'Search'.\n"
|
||||
f"7. If the intent is a bottom navigation tab (e.g. 'profile tab', 'home tab'):\n"
|
||||
f" - These are always at the BOTTOM edge of the screen.\n"
|
||||
f" - 'profile tab' is usually the furthest right icon (your avatar).\n"
|
||||
f" - 'home tab' is the furthest left icon (house).\n"
|
||||
f" - 'explore tab' is the magnifying glass.\n"
|
||||
f" - 'reels tab' is the video clapperboard.\n"
|
||||
f"8. If the intent involves 'author username' or 'author profile':\n"
|
||||
f" - Pick the profile picture (e.g. 'Profile picture of <username>') or the username text.\n"
|
||||
f" - NEVER pick a 'Follow' button. Do NOT pick 'Follow <username>'.\n"
|
||||
f"9. If the intent is 'save post':\n"
|
||||
f" - The save icon is the bookmark icon on the bottom right of the post image/video.\n"
|
||||
f" - Usually has desc='Add to Saved' or 'Save'. Do NOT pick the post text or other action buttons.\n"
|
||||
f"10. DISTINGUISHING BOTTOM TABS vs CONTENT BUTTONS:\n"
|
||||
f" - Bottom Navigation Tabs (Home, Search, Reels, Profile) are ALWAYS at the very bottom (y > 2100).\n"
|
||||
f" - Content Interaction Buttons (Like, Comment, Share, Reactions, Message Input) are attached to posts or threads, NOT the bottom nav bar.\n"
|
||||
f" - If looking for 'message input' or 'type message', do NOT select 'reactions' or emoji icons. Look for an empty text box or 'Message...'.\n"
|
||||
f"11. If the intent is 'feed post content' or 'post media content':\n"
|
||||
f" - Pick the largest box that contains the actual image or video, usually described as 'Photo', 'Video', or 'Carousel'.\n"
|
||||
f"12. If the exact control is NOT visible, return null. Do NOT guess.\n\n"
|
||||
f'Reply ONLY with a valid JSON object: {{"box": <number>}} or {{"box": null}}'
|
||||
)
|
||||
|
||||
@@ -343,8 +724,15 @@ class IntentResolver:
|
||||
use_local_edge=True,
|
||||
images_b64=[annotated_b64],
|
||||
)
|
||||
print(f"DEBUG_INTENT: VLM RAW RESPONSE for '{intent_description}': {res}")
|
||||
data = json.loads(res)
|
||||
box_idx = data.get("box")
|
||||
if box_idx is None:
|
||||
box_idx = data.get("selected_index")
|
||||
if box_idx is None:
|
||||
box_idx = data.get("box_index")
|
||||
if box_idx is None:
|
||||
box_idx = data.get("index")
|
||||
|
||||
if box_idx is not None and box_idx in box_map:
|
||||
selected = box_map[box_idx]
|
||||
@@ -367,7 +755,7 @@ class IntentResolver:
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def _text_based_resolve(
|
||||
self, intent_description: str, candidates: List[SpatialNode], device=None
|
||||
self, intent_description: str, candidates: List[SpatialNode], device=None, screen_height: int = 2400
|
||||
) -> Optional[SpatialNode]:
|
||||
"""
|
||||
Fallback resolution via text descriptions of XML nodes.
|
||||
@@ -379,6 +767,9 @@ class IntentResolver:
|
||||
intent_lower = intent_description.lower()
|
||||
|
||||
filtered_candidates = [n for n in candidates if n.area < 500000]
|
||||
filtered_candidates = self.filter_navigation_conflicts(
|
||||
filtered_candidates, intent_description, screen_height=screen_height
|
||||
)
|
||||
if "profile" in intent_lower:
|
||||
filtered_candidates = [
|
||||
n
|
||||
@@ -394,8 +785,8 @@ class IntentResolver:
|
||||
|
||||
node_context = []
|
||||
for i, node in enumerate(filtered_candidates):
|
||||
text = node.text or ""
|
||||
desc = node.content_desc or ""
|
||||
text = _humanize_desc(node.text or "")
|
||||
desc = _humanize_desc(node.content_desc or "")
|
||||
res_id = node.resource_id or ""
|
||||
node_context.append(f"[{i}] text='{text}', desc='{desc}', id='{res_id}', bounds=[{node.y1},{node.y2}]")
|
||||
|
||||
@@ -403,9 +794,15 @@ class IntentResolver:
|
||||
f"You are a Spatial UI Intent Resolver.\n"
|
||||
f"Goal: Find the single best UI element to interact with to satisfy the intent: '{intent_description}'.\n"
|
||||
f"Candidates:\n" + "\n".join(node_context) + "\n\n"
|
||||
"CRITICAL RULES:\n"
|
||||
"1. If the intent is a bottom navigation tab (e.g. 'profile tab', 'home tab'):\n"
|
||||
" - These are always at the BOTTOM of the screen (typically y > 2100).\n"
|
||||
" - 'profile tab' is usually the furthest right.\n"
|
||||
" - 'home tab' is the furthest left.\n"
|
||||
" - Do NOT select 'Go to <user>'s profile' or other header text.\n"
|
||||
"2. If none of the candidates clearly and safely match the intent, return null.\n\n"
|
||||
"Reply ONLY with a valid JSON object strictly matching this schema:\n"
|
||||
'{"selected_index": <integer or null>}\n'
|
||||
"If none of the candidates match the intent, return null."
|
||||
)
|
||||
|
||||
try:
|
||||
@@ -416,6 +813,7 @@ class IntentResolver:
|
||||
user_prompt=prompt,
|
||||
use_local_edge=True,
|
||||
)
|
||||
print(f"DEBUG_INTENT: TEXT LLM RAW RESPONSE for '{intent_description}': {res}")
|
||||
data = json.loads(res)
|
||||
idx = data.get("selected_index")
|
||||
if idx is not None and 0 <= idx < len(filtered_candidates):
|
||||
|
||||
@@ -43,7 +43,7 @@ class ScreenIdentity:
|
||||
except ImportError:
|
||||
self.screen_memory = None
|
||||
|
||||
def identify(self, xml_dump: str) -> Dict[str, Any]:
|
||||
def identify(self, xml_dump: str, screenshot_b64: str = None) -> Dict[str, Any]:
|
||||
"""
|
||||
Analyzes an XML dump and returns a complete screen description.
|
||||
|
||||
@@ -116,6 +116,11 @@ class ScreenIdentity:
|
||||
}
|
||||
)
|
||||
|
||||
from GramAddict.core.situational_awareness import SituationalAwarenessEngine
|
||||
|
||||
sae = SituationalAwarenessEngine.get_instance()
|
||||
signature = sae._compress_xml(xml_dump) if sae else self._compute_signature(resource_ids, content_descs, texts)
|
||||
|
||||
# ── Foreign app check ──
|
||||
if app_id not in packages:
|
||||
return {
|
||||
@@ -123,18 +128,16 @@ class ScreenIdentity:
|
||||
"available_actions": ["press back", "force start instagram"],
|
||||
"selected_tab": None,
|
||||
"context": {"packages": list(packages)},
|
||||
"signature": self._compute_signature(resource_ids, content_descs, texts),
|
||||
"signature": signature,
|
||||
}
|
||||
|
||||
desc_lower = " ".join(content_descs).lower()
|
||||
text_lower = " ".join(texts).lower()
|
||||
ids_str = " ".join(resource_ids).lower()
|
||||
|
||||
signature = self._compute_signature(resource_ids, content_descs, texts)
|
||||
|
||||
# ── Identify screen type from structural signals ──
|
||||
screen_type = self._classify_screen(
|
||||
resource_ids, content_descs, texts, selected_tab, desc_lower, text_lower, ids_str, signature
|
||||
resource_ids, content_descs, texts, selected_tab, desc_lower, text_lower, ids_str, signature, screenshot_b64
|
||||
)
|
||||
|
||||
# ── Extract available actions from clickable elements ──
|
||||
@@ -153,23 +156,45 @@ class ScreenIdentity:
|
||||
"signature": signature,
|
||||
}
|
||||
|
||||
def _classify_screen(self, ids, descs, texts, selected_tab, desc_lower, text_lower, ids_str, signature=None):
|
||||
"""Classify screen type using Semantic Memory with LLM fallback — NO hardcoded states."""
|
||||
def _classify_screen(
|
||||
self, ids, descs, texts, selected_tab, desc_lower, text_lower, ids_str, signature=None, screenshot_b64=None
|
||||
):
|
||||
"""
|
||||
Classify screen type using Semantic Memory with LLM fallback — NO hardcoded states."""
|
||||
|
||||
# Priority 0: Content-creation overlays that block ALL navigation.
|
||||
# Priority 0: Fetch Qdrant Semantic Cache
|
||||
# We fetch this early to see if there is a 'NORMAL' override for the MODAL check.
|
||||
# We DO NOT let this override deterministic structural heuristics! Fuzzy vector matching
|
||||
# can easily confuse HOME_FEED and OWN_PROFILE if the bottom navigation bar is identical.
|
||||
cached_type_str = None
|
||||
if signature and self.screen_memory and self.screen_memory.is_connected:
|
||||
cached_type_str = self.screen_memory.get_screen_type(signature, similarity_threshold=0.92)
|
||||
|
||||
is_normal_override = cached_type_str == "NORMAL"
|
||||
|
||||
# Priority 1: Content-creation overlays that block ALL navigation.
|
||||
# These full-screen Instagram UIs have no navigation tabs and trap the bot.
|
||||
# Structural detection is O(1), zero LLM calls, and cannot be fooled.
|
||||
creation_flow_markers = ("quick_capture", "gallery_cancel_button", "creation_flow", "reel_camera")
|
||||
if any(marker in ids_str for marker in creation_flow_markers):
|
||||
logger.info("🛡️ [ScreenIdentity] Content-creation overlay detected → MODAL")
|
||||
return ScreenType.MODAL
|
||||
if not is_normal_override:
|
||||
creation_flow_markers = ("quick_capture", "gallery_cancel_button", "creation_flow", "reel_camera")
|
||||
if any(marker in ids_str for marker in creation_flow_markers):
|
||||
logger.info("🛡️ [ScreenIdentity] Content-creation overlay detected → MODAL")
|
||||
return ScreenType.MODAL
|
||||
|
||||
# Priority 1: Structural Heuristics (100% Deterministic)
|
||||
# Priority 2: Structural Heuristics (100% Deterministic)
|
||||
if "unified_follow_list_tab_layout" in ids or "follow_list_container" in ids:
|
||||
return ScreenType.FOLLOW_LIST
|
||||
|
||||
if "profile_header_container" in ids:
|
||||
if selected_tab == "profile_tab":
|
||||
# Profile structural markers
|
||||
PROFILE_MARKERS = (
|
||||
"profile_header_container",
|
||||
"row_profile_header_imageview",
|
||||
"profile_tabs_container",
|
||||
"profile_header_name",
|
||||
)
|
||||
if any(marker in ids for marker in PROFILE_MARKERS):
|
||||
own_profile_texts = ("edit profile", "share profile", "profil bearbeiten", "profil teilen")
|
||||
if selected_tab == "profile_tab" or any(m in desc_lower or m in text_lower for m in own_profile_texts):
|
||||
return ScreenType.OWN_PROFILE
|
||||
return ScreenType.OTHER_PROFILE
|
||||
|
||||
@@ -179,21 +204,19 @@ class ScreenIdentity:
|
||||
if any(marker in ids for marker in REELS_MARKERS):
|
||||
return ScreenType.REELS_FEED
|
||||
|
||||
# DM thread detection — structural markers present inside DM conversations
|
||||
if "direct_thread_header" in ids or "row_thread_composer_edittext" in ids:
|
||||
# DM thread detection — Semantic app-agnostic markers (chat input fields)
|
||||
chat_input_markers = ["Message...", "Nachricht...", "Type a message", "Nachricht senden", "Send a message"]
|
||||
if any(marker in texts for marker in chat_input_markers) or "direct_thread_header" in ids:
|
||||
return ScreenType.DM_THREAD
|
||||
|
||||
# Priority 2: Check Qdrant Semantic Cache (Fuzzy/VLM derived)
|
||||
if signature and self.screen_memory and self.screen_memory.is_connected:
|
||||
cached_type_str = self.screen_memory.get_screen_type(signature, similarity_threshold=0.92)
|
||||
if cached_type_str:
|
||||
try:
|
||||
return ScreenType[cached_type_str]
|
||||
except KeyError:
|
||||
pass
|
||||
|
||||
if "row_feed_button_like" in ids and "row_feed_photo_profile_name" in ids and not selected_tab:
|
||||
return ScreenType.POST_DETAIL
|
||||
# POST_DETAIL vs HOME_FEED: Both have row_feed_* markers. The differentiator
|
||||
# is that HOME_FEED has the main_feed_action_bar (top bar with 'Instagram' title).
|
||||
# POST_DETAIL lacks this because it shows a single expanded post.
|
||||
# Note: We MUST NOT use `not selected_tab` here — posts opened from feed
|
||||
# retain the feed_tab as selected, which previously caused misclassification.
|
||||
if "row_feed_button_like" in ids and "row_feed_photo_profile_name" in ids:
|
||||
if "main_feed_action_bar" not in ids:
|
||||
return ScreenType.POST_DETAIL
|
||||
|
||||
# Story view structural markers — present in full-screen story viewer.
|
||||
# Stories hide the navigation tab bar, so selected_tab is always None.
|
||||
@@ -218,6 +241,8 @@ class ScreenIdentity:
|
||||
return ScreenType.REELS_FEED
|
||||
if selected_tab == "search_tab":
|
||||
return ScreenType.EXPLORE_GRID
|
||||
if "action_bar_search_edit_text" in ids:
|
||||
return ScreenType.EXPLORE_GRID
|
||||
if selected_tab == "profile_tab":
|
||||
return ScreenType.OWN_PROFILE
|
||||
if selected_tab == "direct_tab":
|
||||
@@ -225,41 +250,75 @@ class ScreenIdentity:
|
||||
if "message_input" in ids:
|
||||
return ScreenType.DM_INBOX # Fallback for DM thread as inbox
|
||||
|
||||
# Priority 3: Semantic VLM Classification Fallback
|
||||
# End of structural heuristics
|
||||
|
||||
# Priority 3: Cached Semantic Type (If deterministic heuristics failed)
|
||||
if cached_type_str and cached_type_str != "NORMAL":
|
||||
try:
|
||||
cached_type = ScreenType[cached_type_str]
|
||||
# Enforce absolute structural parity: Story and Reels must have their structural markers.
|
||||
# If they reached Priority 3, it means Priority 2 failed to find their markers.
|
||||
# Therefore, any cache telling us this is a Story/Reel without those markers is hallucinating.
|
||||
if cached_type in (ScreenType.STORY_VIEW, ScreenType.REELS_FEED):
|
||||
logger.warning(
|
||||
f"⚠️ [ScreenIdentity] Rejecting cached {cached_type.name} due to missing structural markers."
|
||||
)
|
||||
else:
|
||||
return cached_type
|
||||
except KeyError:
|
||||
pass
|
||||
|
||||
# Priority 4: Semantic VLM Classification Fallback
|
||||
if not screenshot_b64 and getattr(self, "device", None) is not None:
|
||||
screenshot_b64 = self.device.get_screenshot_b64()
|
||||
|
||||
from GramAddict.core.config import Config
|
||||
from GramAddict.core.llm_provider import query_llm
|
||||
from GramAddict.core.llm_provider import query_telepathic_llm
|
||||
|
||||
cfg = Config()
|
||||
url = (
|
||||
getattr(cfg.args, "ai_embedding_url", "http://localhost:11434/api/chat")
|
||||
getattr(cfg.args, "ai_telepathic_url", "http://localhost:11434/api/generate")
|
||||
if hasattr(cfg, "args")
|
||||
else "http://localhost:11434/api/chat"
|
||||
else "http://localhost:11434/api/generate"
|
||||
)
|
||||
model = getattr(cfg.args, "ai_embedding_model", "llama3") if hasattr(cfg, "args") else "llama3"
|
||||
model = getattr(cfg.args, "ai_telepathic_model", "llava:latest") if hasattr(cfg, "args") else "llava:latest"
|
||||
|
||||
layout_context = (
|
||||
f"Selected Tab: {selected_tab}\nResource IDs: {list(ids)}\nVisible Texts context: {texts[:10]}\n"
|
||||
)
|
||||
prompt = (
|
||||
f"Identify the Instagram screen layout type based on these DOM structural signals.\n"
|
||||
f"Identify the Instagram screen layout type based on the provided screenshot and structural signals.\n"
|
||||
f"Valid types: {[t.name for t in ScreenType]}\n"
|
||||
f"Context:\n{layout_context}\n"
|
||||
f"Reply ONLY with the exact matching enum Type Name string, or 'UNKNOWN' if no type matches."
|
||||
)
|
||||
|
||||
try:
|
||||
response = query_llm(
|
||||
url=url, model=model, prompt="Classify this screen layout.", system=prompt, format_json=False
|
||||
response = query_telepathic_llm(
|
||||
model=model,
|
||||
url=url,
|
||||
system_prompt=prompt,
|
||||
user_prompt="Classify this screen layout.",
|
||||
images_b64=[screenshot_b64] if screenshot_b64 else None,
|
||||
temperature=0.0,
|
||||
use_local_edge=True,
|
||||
)
|
||||
if response and isinstance(response, str):
|
||||
result = response.strip().upper()
|
||||
elif response and isinstance(response, dict) and "response" in response:
|
||||
result = response["response"].strip().upper()
|
||||
else:
|
||||
return ScreenType.UNKNOWN
|
||||
|
||||
result = response.strip().upper() if response else "UNKNOWN"
|
||||
|
||||
for t in ScreenType:
|
||||
if t.name in result:
|
||||
if is_normal_override and t == ScreenType.MODAL:
|
||||
# Prevent the LLM from hallucinating an obstacle if explicitly verified as NORMAL
|
||||
return ScreenType.UNKNOWN
|
||||
|
||||
# Enforce absolute structural parity: Story and Reels must have their structural markers.
|
||||
if t in (ScreenType.STORY_VIEW, ScreenType.REELS_FEED):
|
||||
logger.warning(
|
||||
f"⚠️ [ScreenIdentity] Rejecting VLM hallucinated {t.name} due to missing structural markers."
|
||||
)
|
||||
return ScreenType.UNKNOWN
|
||||
|
||||
if signature and self.screen_memory:
|
||||
self.screen_memory.store_screen(signature, t.name)
|
||||
return t
|
||||
@@ -300,8 +359,19 @@ class ScreenIdentity:
|
||||
actions.append("tap save button")
|
||||
if "back" in desc_lower:
|
||||
actions.append("tap back button")
|
||||
if any("follow" in e.get("text", "").lower() for e in clickable_elements):
|
||||
actions.append("tap 'Follow' button")
|
||||
has_following = any(
|
||||
"following" in e.get("text", "").lower() or "following" in e.get("desc", "").lower()
|
||||
for e in clickable_elements
|
||||
)
|
||||
if has_following:
|
||||
actions.append("tap following button")
|
||||
elif any(
|
||||
"follow" in e.get("text", "").lower()
|
||||
or "follow" in e.get("desc", "").lower()
|
||||
or "follow" in e.get("id", "").lower()
|
||||
for e in clickable_elements
|
||||
):
|
||||
actions.append("tap follow button")
|
||||
|
||||
if screen_type == ScreenType.OWN_PROFILE or screen_type == ScreenType.OTHER_PROFILE:
|
||||
if "message" in desc_lower or "nachricht" in desc_lower:
|
||||
@@ -316,10 +386,11 @@ class ScreenIdentity:
|
||||
|
||||
# Grid items
|
||||
if screen_type == ScreenType.EXPLORE_GRID:
|
||||
actions.append("tap first grid item")
|
||||
actions.append("tap first post")
|
||||
|
||||
# Scroll
|
||||
actions.append("scroll down")
|
||||
actions.append("scroll up")
|
||||
actions.append("press back")
|
||||
|
||||
return list(set(actions)) # Deduplicate
|
||||
|
||||
@@ -117,12 +117,13 @@ class SemanticEvaluator:
|
||||
You are a user with the following interests: {', '.join(persona_interests)}.
|
||||
You are looking at an Instagram post.
|
||||
Evaluate if this post is highly relevant to your interests and if you should like/comment on it.
|
||||
CRITICAL: Check if this post is an advertisement or sponsored content (look for "Sponsored", "Ad", or promotional product placement).
|
||||
|
||||
Reply ONLY in valid JSON format:
|
||||
{{
|
||||
"should_like": true/false,
|
||||
"should_comment": true/false,
|
||||
"reasoning": "brief explanation"
|
||||
"is_ad": true/false
|
||||
}}
|
||||
"""
|
||||
response = self._query_vlm(prompt, screenshot_b64)
|
||||
@@ -131,7 +132,17 @@ class SemanticEvaluator:
|
||||
json_str = response.split("```json")[1].split("```")[0].strip()
|
||||
else:
|
||||
json_str = response.strip()
|
||||
return json.loads(json_str)
|
||||
try:
|
||||
return json.loads(json_str)
|
||||
except json.JSONDecodeError:
|
||||
# Try to close potential unclosed JSON strings
|
||||
if not json_str.endswith("}"):
|
||||
json_str += "}"
|
||||
try:
|
||||
return json.loads(json_str)
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
logger.warning(f"👁️ [Vision Core] VLM returned malformed JSON: {response}")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to evaluate post vibe: {e}")
|
||||
return None
|
||||
|
||||
@@ -184,7 +184,7 @@ class SpatialParser:
|
||||
for n in all_nodes:
|
||||
has_semantic = bool(n.text or n.content_desc)
|
||||
semantic_res = n.resource_id and any(
|
||||
x in n.resource_id.lower() for x in ["button", "tab", "icon", "action", "menu"]
|
||||
x in n.resource_id.lower() for x in ["button", "tab", "icon", "action", "menu", "imageview"]
|
||||
)
|
||||
|
||||
if n.clickable or n.scrollable or semantic_res or (has_semantic and n.area < 500000 and n.area > 0):
|
||||
|
||||
@@ -13,7 +13,8 @@ class PersistentList(list):
|
||||
self.load()
|
||||
|
||||
def load(self):
|
||||
path = f"accounts/{self.filename}.json"
|
||||
base_dir = os.environ.get("GRAMADDICT_ACCOUNTS_DIR", "accounts")
|
||||
path = f"{base_dir}/{self.filename}.json"
|
||||
if os.path.exists(path):
|
||||
try:
|
||||
with open(path, "r") as f:
|
||||
@@ -27,9 +28,8 @@ class PersistentList(list):
|
||||
self.persist()
|
||||
|
||||
def persist(self, directory=None):
|
||||
if os.environ.get("PYTEST_CURRENT_TEST"):
|
||||
return
|
||||
folder = f"accounts/{directory}" if directory else "accounts"
|
||||
base_dir = os.environ.get("GRAMADDICT_ACCOUNTS_DIR", "accounts")
|
||||
folder = f"{base_dir}/{directory}" if directory else base_dir
|
||||
os.makedirs(folder, exist_ok=True)
|
||||
path = f"{folder}/{self.filename}.json"
|
||||
try:
|
||||
|
||||
@@ -136,11 +136,12 @@ def align_active_post(device):
|
||||
aligned = False
|
||||
attempts = 0
|
||||
max_attempts = 5 # Increased for structural retry loop
|
||||
failed_bounds = set()
|
||||
|
||||
# Intents for structural discovery
|
||||
intents = [
|
||||
"post author username text (exclude follow buttons)",
|
||||
"post author header profile",
|
||||
"post username name",
|
||||
"row_feed_photo_profile_name", # ID fallback
|
||||
"clips_viewer_author_container", # Reels fallback
|
||||
"feed post content", # Final desperation
|
||||
@@ -150,13 +151,19 @@ def align_active_post(device):
|
||||
attempts += 1
|
||||
try:
|
||||
xml = device.dump_hierarchy()
|
||||
if "clips_video_container" in xml or "clips_viewer_container" in xml:
|
||||
logger.info("🎯 [Alignment] Reels view detected. Auto-snapping is native.")
|
||||
return True
|
||||
|
||||
from GramAddict.core.telepathic_engine import TelepathicEngine
|
||||
|
||||
telepath = TelepathicEngine.get_instance()
|
||||
|
||||
target_node = None
|
||||
for intent in intents:
|
||||
target_node = telepath.find_best_node(xml, intent, min_confidence=0.35, device=device, track=False)
|
||||
target_node = telepath.find_best_node(
|
||||
xml, intent, min_confidence=0.35, device=device, track=False, exclude_bounds=list(failed_bounds)
|
||||
)
|
||||
if target_node:
|
||||
break
|
||||
|
||||
@@ -164,9 +171,11 @@ def align_active_post(device):
|
||||
original_attribs = target_node.get("original_attribs", {})
|
||||
bounds = original_attribs.get("bounds")
|
||||
|
||||
bounds_str = ""
|
||||
# If bounds is a tuple from SpatialNode.to_dict()
|
||||
if isinstance(bounds, (tuple, list)) and len(bounds) == 4:
|
||||
left, t, r, b = bounds
|
||||
bounds_str = f"[{left},{t}][{r},{b}]"
|
||||
else:
|
||||
# Fallback to string parsing
|
||||
if not bounds:
|
||||
@@ -174,6 +183,7 @@ def align_active_post(device):
|
||||
m = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", str(bounds))
|
||||
if m:
|
||||
left, t, r, b = map(int, m.groups())
|
||||
bounds_str = f"[{left},{t}][{r},{b}]"
|
||||
else:
|
||||
logger.warning(f"📐 [Alignment] Could not parse bounds: {bounds}")
|
||||
continue
|
||||
@@ -184,6 +194,7 @@ def align_active_post(device):
|
||||
h = info.get("displayHeight", 2400)
|
||||
if t > h * 0.85:
|
||||
logger.debug(f"📐 [Alignment] Rejecting node at y={t} (too low, likely bottom bar)")
|
||||
failed_bounds.add(bounds_str)
|
||||
continue
|
||||
|
||||
header_y = (t + b) // 2
|
||||
|
||||
@@ -121,32 +121,11 @@ class QNavGraph:
|
||||
GOAP-powered action execution.
|
||||
Replaces _execute_transition() for post interactions.
|
||||
|
||||
Screen-aware: refuses to attempt actions that don't exist on the current screen.
|
||||
|
||||
Usage:
|
||||
nav_graph.do("like this post") # instead of _execute_transition("tap_like_button")
|
||||
nav_graph.do("follow this user") # instead of _execute_transition("tap_follow_button")
|
||||
nav_graph.do("tap first grid item") # instead of _execute_transition("tap_explore_grid_item")
|
||||
"""
|
||||
# ── Screen sanity check: is this action possible here? ──
|
||||
screen = self.goap.perceive()
|
||||
available = screen.get("available_actions", [])
|
||||
screen_type = screen["screen_type"]
|
||||
|
||||
# Map goal to the action that should be available
|
||||
action_checks = {
|
||||
"like": "tap like button",
|
||||
"comment": "tap comment button",
|
||||
"share": "tap share button",
|
||||
"follow": "tap follow button",
|
||||
}
|
||||
for keyword, required_action in action_checks.items():
|
||||
if keyword in goal.lower() and required_action not in available:
|
||||
logger.warning(
|
||||
f"🚫 [GOAP] Cannot '{goal}' on {screen_type.value} "
|
||||
f"('{required_action}' not available on this screen)"
|
||||
)
|
||||
return False
|
||||
|
||||
return self.goap._execute_action(goal)
|
||||
|
||||
@@ -172,13 +151,13 @@ class QNavGraph:
|
||||
success = self.sae.ensure_clear_screen(max_attempts=max_attempts + 5, initial_xml=xml_dump)
|
||||
return success
|
||||
|
||||
def _execute_transition(self, action: str, mock_semantic_engine=None, max_retries: int = 2) -> bool:
|
||||
def _execute_transition(self, action: str, max_retries: int = 2) -> bool:
|
||||
"""
|
||||
Executes a transition (e.g. 'tap_explore_tab') using the Telepathic Semantic Engine.
|
||||
"""
|
||||
from GramAddict.core.telepathic_engine import TelepathicEngine
|
||||
|
||||
engine = mock_semantic_engine or TelepathicEngine.get_instance()
|
||||
engine = TelepathicEngine.get_instance()
|
||||
|
||||
failed_positions = set() # Track (x, y) of clicks that failed, for grid retry diversity
|
||||
|
||||
|
||||
@@ -141,7 +141,7 @@ class QdrantBase:
|
||||
url,
|
||||
json=payload,
|
||||
headers=headers,
|
||||
timeout=12,
|
||||
timeout=30,
|
||||
)
|
||||
if resp.status_code != 200:
|
||||
logger.debug(f"Embedding API Error {resp.status_code}: {resp.text}")
|
||||
@@ -184,6 +184,7 @@ class QdrantBase:
|
||||
self.client.upsert(
|
||||
collection_name=self.collection_name,
|
||||
points=[PointStruct(id=point_id, vector=safe_vector, payload=payload)],
|
||||
wait=True,
|
||||
)
|
||||
|
||||
# ABSOLUTE LOGGING: User requirement for full observability
|
||||
@@ -429,8 +430,9 @@ class UIMemoryDB(QdrantBase):
|
||||
if exact_points:
|
||||
eval_result = _evaluate_payload(exact_points[0].payload, score=1.0, point_id=point_id)
|
||||
if eval_result:
|
||||
logger.debug(
|
||||
f"Resolved intent '{intent}' from Qdrant Memory via EXACT ID MATCH! (Confidence: {eval_result['effective_confidence']:.2f})"
|
||||
logger.info(
|
||||
f"🧠 [Memory] Applying learned pattern for '{intent}' (EXACT MATCH, Confidence: {eval_result['effective_confidence']:.2f})",
|
||||
extra={"color": "\x1b[36m"}, # Cyan color
|
||||
)
|
||||
return eval_result["solution"]
|
||||
# If exact match failed evaluation (e.g. decayed), we shouldn't fall back to vector search because it's the exact intent!
|
||||
@@ -459,8 +461,9 @@ class UIMemoryDB(QdrantBase):
|
||||
if results and results[0].score >= similarity_threshold:
|
||||
eval_result = _evaluate_payload(results[0].payload, score=results[0].score, point_id=results[0].id)
|
||||
if eval_result:
|
||||
logger.debug(
|
||||
f"Resolved intent '{intent}' from Qdrant Memory via vector search! (Score: {results[0].score:.3f}, Confidence: {eval_result['effective_confidence']:.2f})"
|
||||
logger.info(
|
||||
f"🧠 [Memory] Applying learned pattern for '{intent}' (VECTOR MATCH, Score: {results[0].score:.3f}, Confidence: {eval_result['effective_confidence']:.2f})",
|
||||
extra={"color": "\x1b[36m"}, # Cyan color
|
||||
)
|
||||
return eval_result["solution"]
|
||||
return None
|
||||
@@ -511,7 +514,10 @@ class UIMemoryDB(QdrantBase):
|
||||
],
|
||||
wait=True,
|
||||
)
|
||||
logger.info(f"Learned pattern for '{intent}' and saved to Qdrant Memory (ID: {point_id[:8]}...).")
|
||||
logger.info(
|
||||
f"📥 [Memory] Learned new pattern for '{intent}' and saved to Qdrant (ID: {point_id[:8]}...)",
|
||||
extra={"color": "\x1b[35m"}, # Magenta color
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"Qdrant storage error: {e}")
|
||||
|
||||
@@ -573,7 +579,12 @@ class UIMemoryDB(QdrantBase):
|
||||
payload={"confidence": new_confidence},
|
||||
points=[point_id],
|
||||
)
|
||||
logger.debug(f"Confidence for '{intent}' adjusted to {new_confidence:.2f} (delta: {delta:+.2f}).")
|
||||
color = "\x1b[32m" if delta > 0 else "\x1b[31m" # Green for positive, Red for negative
|
||||
symbol = "📈 [Memory] Positive Reinforcement:" if delta > 0 else "📉 [Memory] Negative Reinforcement:"
|
||||
logger.info(
|
||||
f"{symbol} Confidence for '{intent}' adjusted to {new_confidence:.2f} (delta: {delta:+.2f})",
|
||||
extra={"color": color},
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f"Confidence adjustment error: {e}")
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import logging
|
||||
import math
|
||||
import random
|
||||
from typing import Optional
|
||||
|
||||
from colorama import Fore
|
||||
@@ -331,14 +332,32 @@ class ResonanceEngine:
|
||||
is_comment_node = "comment" in res_id or "textview" in res_id
|
||||
|
||||
# 3. Block accessibility garbage & UI labels
|
||||
# Zero-Maintenance: Only structural patterns. Short strings
|
||||
# (< 5 chars) from UI buttons are blocked by length, not by
|
||||
# translating every possible language.
|
||||
is_ui_junk = (
|
||||
val.lower().startswith("go to")
|
||||
or val.lower().startswith("tap to")
|
||||
or "actions for this post" in val.lower()
|
||||
or len(val.strip()) < 3
|
||||
)
|
||||
|
||||
# Block known English UI action labels.
|
||||
# We intentionally do NOT add German/Spanish/etc translations.
|
||||
# Instead, we rely on the structural `is_comment_node` filter
|
||||
# above + length heuristic to catch non-comment UI elements.
|
||||
blocked_exact = [
|
||||
"reply",
|
||||
"like",
|
||||
"view replies",
|
||||
"see translation",
|
||||
"hide replies",
|
||||
"view all comments",
|
||||
"send",
|
||||
]
|
||||
|
||||
if val and len(val) > 2 and is_comment_node and not is_ui_junk:
|
||||
if val.lower() not in ["reply", "like", "view replies", "see translation", "hide replies"]:
|
||||
if val.lower() not in blocked_exact:
|
||||
raw_comments.append(val)
|
||||
except Exception as e:
|
||||
logger.error(f"🧠 [Comment Learning] Failed to parse XML: {e}")
|
||||
@@ -393,7 +412,7 @@ class ResonanceEngine:
|
||||
logger.debug(f"DEBUG CONDENSER RAW: {response_text}")
|
||||
|
||||
# Parse json gracefully
|
||||
if type(response_text) is str:
|
||||
if isinstance(response_text, str):
|
||||
clean_json = response_text.strip()
|
||||
if clean_json.startswith("```json"):
|
||||
clean_json = clean_json[7:]
|
||||
|
||||
@@ -33,6 +33,7 @@ class ScreenTopology:
|
||||
"tap profile tab": ScreenType.OWN_PROFILE,
|
||||
"tap reels tab": ScreenType.REELS_FEED,
|
||||
"tap messages tab": ScreenType.DM_INBOX,
|
||||
"tap story ring avatar": ScreenType.STORY_VIEW,
|
||||
},
|
||||
ScreenType.EXPLORE_GRID: {
|
||||
"tap home tab": ScreenType.HOME_FEED,
|
||||
@@ -57,9 +58,17 @@ class ScreenTopology:
|
||||
ScreenType.FOLLOW_LIST: {
|
||||
"press back": ScreenType.OWN_PROFILE,
|
||||
},
|
||||
ScreenType.STORY_VIEW: {
|
||||
"press back": ScreenType.HOME_FEED,
|
||||
},
|
||||
ScreenType.OTHER_PROFILE: {
|
||||
"press back": ScreenType.HOME_FEED,
|
||||
},
|
||||
ScreenType.POST_DETAIL: {
|
||||
"tap home tab": ScreenType.HOME_FEED,
|
||||
"tap explore tab": ScreenType.EXPLORE_GRID,
|
||||
"tap profile tab": ScreenType.OWN_PROFILE,
|
||||
"tap reels tab": ScreenType.REELS_FEED,
|
||||
},
|
||||
ScreenType.UNKNOWN: {
|
||||
"tap home tab": ScreenType.HOME_FEED,
|
||||
@@ -88,7 +97,9 @@ class ScreenTopology:
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def find_route(cls, from_screen: ScreenType, to_screen: ScreenType, avoid_actions: set = None) -> Optional[List[Tuple[str, ScreenType]]]:
|
||||
def find_route(
|
||||
cls, from_screen: ScreenType, to_screen: ScreenType, avoid_actions: set = None
|
||||
) -> Optional[List[Tuple[str, ScreenType]]]:
|
||||
"""
|
||||
BFS shortest path from from_screen to to_screen.
|
||||
|
||||
@@ -99,7 +110,7 @@ class ScreenTopology:
|
||||
"""
|
||||
if from_screen == to_screen:
|
||||
return []
|
||||
|
||||
|
||||
avoid_actions = avoid_actions or set()
|
||||
|
||||
queue: deque = deque()
|
||||
@@ -113,7 +124,7 @@ class ScreenTopology:
|
||||
for action, next_screen in transitions.items():
|
||||
if action in avoid_actions or action.replace(" ", "_") in avoid_actions:
|
||||
continue
|
||||
|
||||
|
||||
if next_screen == to_screen:
|
||||
return path + [(action, next_screen)]
|
||||
|
||||
|
||||
@@ -277,7 +277,29 @@ class SessionState:
|
||||
|
||||
|
||||
class SessionStateEncoder(JSONEncoder):
|
||||
"""JSON encoder for SessionState that is crash-proof against non-serializable types."""
|
||||
|
||||
_SAFE_TYPES = (str, int, float, bool, type(None))
|
||||
|
||||
@classmethod
|
||||
def _sanitize_value(cls, value):
|
||||
"""Convert any non-JSON-serializable value to a safe string representation."""
|
||||
if isinstance(value, cls._SAFE_TYPES):
|
||||
return value
|
||||
if isinstance(value, datetime):
|
||||
return value.isoformat()
|
||||
if isinstance(value, dict):
|
||||
return {k: cls._sanitize_value(v) for k, v in value.items()}
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [cls._sanitize_value(v) for v in value]
|
||||
# Last resort: stringify unknown objects to prevent json.dump mid-write crashes
|
||||
return str(value)
|
||||
|
||||
def default(self, session_state: SessionState):
|
||||
# Sanitize args dict — never trust raw __dict__, it may contain datetime or other garbage
|
||||
raw_args = session_state.args.__dict__ if hasattr(session_state.args, "__dict__") else {}
|
||||
safe_args = {k: self._sanitize_value(v) for k, v in raw_args.items()}
|
||||
|
||||
return {
|
||||
"id": session_state.id,
|
||||
"total_interactions": sum(session_state.totalInteractions.values()),
|
||||
@@ -291,7 +313,7 @@ class SessionStateEncoder(JSONEncoder):
|
||||
"total_scraped": session_state.totalScraped,
|
||||
"start_time": str(session_state.startTime),
|
||||
"finish_time": str(session_state.finishTime),
|
||||
"args": session_state.args.__dict__,
|
||||
"args": safe_args,
|
||||
"profile": {
|
||||
"posts": session_state.my_posts_count,
|
||||
"followers": session_state.my_followers_count,
|
||||
|
||||
@@ -270,7 +270,13 @@ class SituationalAwarenessEngine:
|
||||
if clickable == "true":
|
||||
parts.append("CLICKABLE")
|
||||
if bounds:
|
||||
parts.append(f"bounds={bounds}")
|
||||
nums = [int(n) for n in re.findall(r"\d+", bounds)]
|
||||
if len(nums) == 4:
|
||||
cx = (nums[0] + nums[2]) // 2
|
||||
cy = (nums[1] + nums[3]) // 2
|
||||
parts.append(f"bounds={bounds} center=({cx},{cy})")
|
||||
else:
|
||||
parts.append(f"bounds={bounds}")
|
||||
|
||||
elements.append(" | ".join(parts))
|
||||
|
||||
@@ -293,8 +299,6 @@ class SituationalAwarenessEngine:
|
||||
if not xml_dump or not isinstance(xml_dump, str):
|
||||
return SituationType.OBSTACLE_FOREIGN_APP
|
||||
|
||||
xml_dump.lower()
|
||||
|
||||
blocked_markers = [
|
||||
"try again later",
|
||||
"action blocked",
|
||||
@@ -346,8 +350,31 @@ class SituationalAwarenessEngine:
|
||||
is_foreign = True
|
||||
|
||||
if is_foreign:
|
||||
# We explicitly ask the TelepathicEngine to classify this to avoid writing brittle substring hacks
|
||||
# for Android System UI variations across different device manufacturers.
|
||||
# ── Tier 1: Known Foreign Packages (O(1) — ZERO LLM) ──
|
||||
# Production bug 2026-05-03: Play Store was detected via slow LLM path.
|
||||
# For these well-known packages, a set lookup is instant and infallible.
|
||||
KNOWN_FOREIGN_PACKAGES = {
|
||||
"com.android.vending", # Play Store
|
||||
"com.android.chrome", # Chrome
|
||||
"com.google.android.chrome", # Chrome (Google build)
|
||||
"com.google.android.youtube", # YouTube
|
||||
"org.mozilla.firefox", # Firefox
|
||||
"com.opera.browser", # Opera
|
||||
"com.brave.browser", # Brave
|
||||
"com.microsoft.emmx", # Edge
|
||||
"com.sec.android.app.sbrowser", # Samsung Browser
|
||||
}
|
||||
dominant_pkgs = packages - {"com.android.systemui"}
|
||||
fast_match = dominant_pkgs & KNOWN_FOREIGN_PACKAGES
|
||||
if fast_match:
|
||||
logger.info(
|
||||
f"🚨 [SAE Perceive] Known foreign package: {fast_match} → "
|
||||
f"OBSTACLE_FOREIGN_APP (O(1) fast-path, no LLM needed)"
|
||||
)
|
||||
return SituationType.OBSTACLE_FOREIGN_APP
|
||||
|
||||
# ── Tier 2: Unknown/Ambiguous Packages → LLM Classification ──
|
||||
# Only SystemUI-only or rare custom packages reach this path.
|
||||
try:
|
||||
from GramAddict.core.config import Config
|
||||
from GramAddict.core.llm_provider import query_telepathic_llm
|
||||
@@ -406,11 +433,21 @@ class SituationalAwarenessEngine:
|
||||
|
||||
compressed = self._compress_xml(xml_dump)
|
||||
|
||||
cached_type = screen_memory.get_screen_type(compressed)
|
||||
|
||||
if cached_type:
|
||||
if cached_type == "OBSTACLE_MODAL":
|
||||
return SituationType.OBSTACLE_MODAL
|
||||
elif cached_type == "NORMAL":
|
||||
return SituationType.NORMAL
|
||||
|
||||
# ── Structural Fast-Check: Content-Creation Overlays ──
|
||||
# These full-screen overlays live INSIDE Instagram's package but block
|
||||
# all normal navigation. They are invisible to the foreign-app detector
|
||||
# and frequently fool the LLM into thinking they are "normal" browsing.
|
||||
# Detecting them structurally is O(1) and requires ZERO LLM calls.
|
||||
# This is checked AFTER Qdrant to ensure that if the LLM unlearned a false positive,
|
||||
# we respect the learned NORMAL state and don't infinite-loop.
|
||||
creation_flow_markers = (
|
||||
"quick_capture", # Camera / story capture overlay
|
||||
"gallery_cancel_button", # Story gallery "Back to Home" button
|
||||
@@ -427,13 +464,56 @@ class SituationalAwarenessEngine:
|
||||
screen_memory.store_screen(compressed, "OBSTACLE_MODAL")
|
||||
return SituationType.OBSTACLE_MODAL
|
||||
|
||||
cached_type = screen_memory.get_screen_type(compressed)
|
||||
# ── Structural Fast-Check: Instagram-Internal Modal Overlays ──
|
||||
# Surveys, rating prompts, and interstitial modals live INSIDE Instagram's
|
||||
# package but block normal interaction. They share a common structural
|
||||
# pattern: a container resource-id containing "survey", "interstitial",
|
||||
# or "nux_" (new-user-experience), plus dismiss buttons ("Not Now").
|
||||
# Detecting them structurally is O(1) and eliminates LLM hallucination risk.
|
||||
instagram_modal_markers = (
|
||||
"survey_overlay_container", # "How are you enjoying Instagram?" survey
|
||||
"survey_title", # Survey title text view
|
||||
"interstitial_container", # Generic interstitial blocker
|
||||
"mystery_interstitial", # Unknown/dynamic interstitials
|
||||
"nux_overlay", # New-user-experience onboarding modals
|
||||
"rating_prompt", # App Store rating prompt
|
||||
"feedback_dialog", # Feedback collection dialogs
|
||||
)
|
||||
if any(
|
||||
re.search(rf'resource-id="[^"]*{marker}[^"]*"', xml_dump, re.IGNORECASE)
|
||||
for marker in instagram_modal_markers
|
||||
):
|
||||
logger.info("🧠 [SAE Perceive] Instagram modal overlay detected structurally → OBSTACLE_MODAL")
|
||||
screen_memory.store_screen(compressed, "OBSTACLE_MODAL")
|
||||
return SituationType.OBSTACLE_MODAL
|
||||
|
||||
if cached_type:
|
||||
if cached_type == "OBSTACLE_MODAL":
|
||||
# Fallback heuristic: detect modals by dismiss-button text patterns.
|
||||
# If we see "Not Now" or "Take Survey" as button text inside Instagram, it's a modal.
|
||||
# Guard: match ONLY inside short text attributes (< 40 chars) to avoid caption false positives.
|
||||
dismiss_button_patterns = (
|
||||
r'text="Not Now"',
|
||||
r'text="not now"',
|
||||
r'text="Nicht jetzt"', # German: "Not Now"
|
||||
r'text="Take Survey"',
|
||||
r'text="rate \d+ stars?"', # "rate 5 stars"
|
||||
r'text="Bewerten"', # German: "Rate"
|
||||
)
|
||||
has_dismiss_button = any(re.search(p, xml_dump, re.IGNORECASE) for p in dismiss_button_patterns)
|
||||
if has_dismiss_button:
|
||||
# Cross-validate: must also have a container that looks like a dialog/overlay
|
||||
# (not just a random "Not Now" text in a DM thread or post caption)
|
||||
has_overlay_structure = bool(
|
||||
re.search(
|
||||
r'resource-id="[^"]*(?:overlay|dialog|interstitial|survey|sheet|prompt)[^"]*"',
|
||||
xml_dump,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
or re.search(r'resource-id="[^"]*button_(?:negative|positive)[^"]*"', xml_dump, re.IGNORECASE)
|
||||
)
|
||||
if has_overlay_structure:
|
||||
logger.info("🧠 [SAE Perceive] Instagram dismiss-button modal detected structurally → OBSTACLE_MODAL")
|
||||
screen_memory.store_screen(compressed, "OBSTACLE_MODAL")
|
||||
return SituationType.OBSTACLE_MODAL
|
||||
elif cached_type == "NORMAL":
|
||||
return SituationType.NORMAL
|
||||
|
||||
# If not cached, query LLM for autonomous structural classification
|
||||
try:
|
||||
@@ -442,7 +522,7 @@ class SituationalAwarenessEngine:
|
||||
|
||||
prompt = (
|
||||
"You are a Situation Classifier for a mobile automation agent.\n"
|
||||
"Analyze the given Android UI XML dump. Is there a blocking MODAL, DIALOG, or POPUP "
|
||||
"Analyze the given Android UI XML dump AND screenshot. Is there a blocking MODAL, DIALOG, or POPUP "
|
||||
"covering the screen that needs to be dismissed, or is this a NORMAL usable screen?\n"
|
||||
"A 'clean_sheet_container' with standard Instagram feed content is NORMAL.\n"
|
||||
"A survey, rating prompt, 'not now' prompt, or permission dialog is an OBSTACLE_MODAL.\n"
|
||||
@@ -459,11 +539,17 @@ class SituationalAwarenessEngine:
|
||||
args = Config().args
|
||||
except Exception:
|
||||
pass
|
||||
model = getattr(args, "ai_model", "qwen3.5:latest")
|
||||
url = getattr(args, "ai_model_url", "http://localhost:11434/api/generate")
|
||||
model = getattr(args, "ai_telepathic_model", "llava:latest")
|
||||
url = getattr(args, "ai_telepathic_url", "http://localhost:11434/api/generate")
|
||||
|
||||
screenshot_b64 = getattr(self.device, "get_screenshot_b64", lambda: None)()
|
||||
res = query_telepathic_llm(
|
||||
model=model, url=url, system_prompt="Strict JSON classifier.", user_prompt=prompt, use_local_edge=True
|
||||
model=model,
|
||||
url=url,
|
||||
system_prompt="Strict JSON classifier.",
|
||||
user_prompt=prompt,
|
||||
images_b64=[screenshot_b64] if screenshot_b64 else None,
|
||||
use_local_edge=True,
|
||||
)
|
||||
import json
|
||||
|
||||
@@ -504,27 +590,31 @@ class SituationalAwarenessEngine:
|
||||
Called ONLY when recall AND structural planning both miss.
|
||||
"""
|
||||
from GramAddict.core.config import Config
|
||||
from GramAddict.core.llm_provider import query_llm
|
||||
from GramAddict.core.llm_provider import query_telepathic_llm
|
||||
|
||||
try:
|
||||
args = Config().args
|
||||
model = getattr(args, "ai_fallback_model", "llama3.2:1b")
|
||||
url = getattr(args, "ai_fallback_url", "http://localhost:11434/api/generate")
|
||||
model = getattr(args, "ai_telepathic_model", "llava:latest")
|
||||
url = getattr(args, "ai_telepathic_url", "http://localhost:11434/api/generate")
|
||||
except Exception:
|
||||
model = "llama3.2:1b"
|
||||
model = "llava:latest"
|
||||
url = "http://localhost:11434/api/generate"
|
||||
|
||||
system_prompt = (
|
||||
"You are an Android UI navigation agent. Your job is to escape obstacles "
|
||||
"(dialogs, modals, foreign apps, system popups) and return to Instagram. "
|
||||
"Analyze the screen content and return a JSON escape action.\n\n"
|
||||
"Analyze the screen content (Screenshot AND XML) and return a JSON escape action.\n\n"
|
||||
"Rules:\n"
|
||||
"- If you see a dismiss/close/cancel/skip/not now button, click it\n"
|
||||
"- If the Situation type is OBSTACLE_LOCKED_SCREEN, action must be 'unlock'\n"
|
||||
"- If the Situation type is OBSTACLE_FOREIGN_APP, action must be 'kill_foreign_apps'\n"
|
||||
"- If the Situation type is obstacle_locked_screen, action must be 'unlock'\n"
|
||||
"- If the Situation type is obstacle_foreign_app, action must be 'kill_foreign_apps'\n"
|
||||
"- If the Situation type is obstacle_system, you MUST look for 'Deny', 'Don't allow', or 'Cancel' and click it. \n"
|
||||
" NEVER click 'Allow', 'OK', or 'Confirm' on system permissions.\n"
|
||||
" If no negative action button exists, action must be 'back'\n"
|
||||
"- If there is NO obstacle and the screen is a normal Instagram view (false positive), action must be 'false_positive'\n"
|
||||
"- If nothing else works, suggest 'app_start' to force-reopen Instagram\n"
|
||||
"- NEVER click 'OK'/'Confirm'/'Accept' on surveys or prompts\n"
|
||||
"- When you choose to click, you MUST use the EXACT coordinates provided in `center=(x,y)` for that element in the XML\n"
|
||||
'- Return ONLY valid JSON: {"action": "click"|"back"|"app_start"|"unlock"|"kill_foreign_apps"|"false_positive", "x": N, "y": N, "reason": "..."}'
|
||||
)
|
||||
|
||||
@@ -535,20 +625,31 @@ class SituationalAwarenessEngine:
|
||||
user_prompt += "What action should I take to clear this obstacle and return to Instagram? Return JSON only."
|
||||
|
||||
try:
|
||||
resp = query_llm(
|
||||
screenshot_b64 = getattr(self.device, "get_screenshot_b64", lambda: None)()
|
||||
|
||||
resp = query_telepathic_llm(
|
||||
url=url,
|
||||
model=model,
|
||||
prompt=user_prompt,
|
||||
system=system_prompt,
|
||||
format_json=True,
|
||||
timeout=30,
|
||||
max_tokens=300,
|
||||
user_prompt=user_prompt,
|
||||
system_prompt=system_prompt,
|
||||
images_b64=[screenshot_b64] if screenshot_b64 else None,
|
||||
temperature=0.0,
|
||||
)
|
||||
if resp and "response" in resp:
|
||||
if resp:
|
||||
import json
|
||||
|
||||
data = json.loads(resp["response"])
|
||||
try:
|
||||
data = json.loads(resp)
|
||||
except json.JSONDecodeError:
|
||||
# Try extracting JSON via regex if LLM was chatty
|
||||
import re
|
||||
|
||||
match = re.search(r"\{.*\}", resp, re.DOTALL)
|
||||
if match:
|
||||
data = json.loads(match.group(0))
|
||||
else:
|
||||
raise ValueError(f"Could not parse JSON from: {resp}")
|
||||
|
||||
return EscapeAction(
|
||||
action_type=data.get("action", "back"),
|
||||
x=int(data.get("x", 0)),
|
||||
@@ -660,6 +761,24 @@ class SituationalAwarenessEngine:
|
||||
|
||||
logger.warning(f"🔍 [SAE] Obstacle detected: {situation.value} (attempt {attempt + 1}/{max_attempts})")
|
||||
|
||||
# ── O(1) Fast-Path for Foreign Apps ──
|
||||
if situation == SituationType.OBSTACLE_FOREIGN_APP:
|
||||
logger.warning("⚡ [SAE Fast-Path] Foreign App detected. Bypassing LLM and killing immediately.")
|
||||
action = EscapeAction("kill_foreign_apps", reason="O(1) fast-path to eliminate foreign app")
|
||||
self._execute_escape(action)
|
||||
|
||||
# Check if we recovered
|
||||
post_xml = self.device.dump_hierarchy()
|
||||
if self.perceive(post_xml) == SituationType.NORMAL:
|
||||
logger.info("✅ [SAE Fast-Path] Foreign App cleared successfully!")
|
||||
self._consecutive_failures = 0
|
||||
return True
|
||||
|
||||
# If we didn't recover, log it and let the loop continue
|
||||
logger.warning("⚠️ [SAE Fast-Path] kill_foreign_apps did not return to NORMAL. Retrying...")
|
||||
self._consecutive_failures += 1
|
||||
continue
|
||||
|
||||
# ── COMPRESS for memory lookup ──
|
||||
compressed = self._compress_xml(xml_dump)
|
||||
|
||||
|
||||
@@ -54,22 +54,20 @@ class TelepathicEngine:
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def find_best_node(
|
||||
self, xml_string: str, intent_description: str, device=None, track: bool = True, **kwargs
|
||||
self,
|
||||
xml_string: str,
|
||||
intent_description: str,
|
||||
device=None,
|
||||
track: bool = True,
|
||||
exclude_bounds: list[str] = None,
|
||||
**kwargs,
|
||||
) -> Optional[dict]:
|
||||
print("FIND_BEST_NODE CALLED")
|
||||
|
||||
"""
|
||||
Public facade for resolving a node.
|
||||
Translates Android UI bounds into standard GramAddict node dicts.
|
||||
"""
|
||||
logger.debug(f"🧠 [SpatialEngine] Resolving intent: '{intent_description}'")
|
||||
|
||||
# 1.25 Structural Fast-Paths (Deterministically bypass VLM for fixed UI elements)
|
||||
nodes_dicts = self._extract_semantic_nodes(xml_string)
|
||||
fast_node = self._structural_fast_path(intent_description, nodes_dicts, kwargs.get("skip_positions"), xml_string)
|
||||
if fast_node:
|
||||
return fast_node
|
||||
|
||||
# 1. Parse into Spatial Topology
|
||||
root = self._parser.parse(xml_string)
|
||||
if not root:
|
||||
@@ -79,6 +77,14 @@ class TelepathicEngine:
|
||||
# 2. Extract interactable candidates
|
||||
candidates = self._parser.get_clickable_nodes(root)
|
||||
|
||||
if exclude_bounds:
|
||||
filtered_candidates = []
|
||||
for c in candidates:
|
||||
bounds_str = f"[{c.x1},{c.y1}][{c.x2},{c.y2}]"
|
||||
if bounds_str not in exclude_bounds:
|
||||
filtered_candidates.append(c)
|
||||
candidates = filtered_candidates
|
||||
|
||||
# 3. Resolve intent against candidates
|
||||
best_node = self._resolver.resolve(intent_description, candidates, device=device)
|
||||
|
||||
@@ -86,13 +92,28 @@ class TelepathicEngine:
|
||||
logger.warning(f"No viable nodes found for intent: '{intent_description}'")
|
||||
return None
|
||||
|
||||
# 3.1 BUG 7 Fix: Semantic Guard for 'post media content'
|
||||
intent_lower = intent_description.lower()
|
||||
semantic_str = (
|
||||
(best_node.text or "") + " " + (best_node.content_desc or "") + " " + (best_node.resource_id or "")
|
||||
).lower()
|
||||
if "post media content" in intent_lower:
|
||||
if "follow" in semantic_str.replace("_", " "):
|
||||
logger.warning("🚫 [SpatialEngine] VLM selected a 'Follow' button for 'post media content'. Blocked.")
|
||||
return None
|
||||
|
||||
# 3.5 Following Button Guard
|
||||
if "follow" in intent_description.lower() and "unfollow" not in intent_description.lower():
|
||||
if (
|
||||
"follow" in intent_description.lower()
|
||||
and "unfollow" not in intent_description.lower()
|
||||
and "following" not in intent_description.lower()
|
||||
):
|
||||
semantic = (
|
||||
(best_node.text or "") + " " + (best_node.content_desc or "") + " " + (best_node.resource_id or "")
|
||||
)
|
||||
semantic = semantic.lower()
|
||||
if "following" in semantic or "gefolgt" in semantic or "requested" in semantic or "angefragt" in semantic:
|
||||
# Zero-Maintenance: Only English UI states. resource_id never changes with locale.
|
||||
if "following" in semantic or "requested" in semantic:
|
||||
return {"skip": True, "semantic": "already_followed"}
|
||||
|
||||
# 4. Track action
|
||||
@@ -134,116 +155,6 @@ class TelepathicEngine:
|
||||
nodes = self._parser.get_clickable_nodes(root)
|
||||
return [self._translate_node(n) for n in nodes]
|
||||
|
||||
def _structural_fast_path(self, intent_description: str, nodes: list, skip_positions: set = None, xml_string: str = "") -> Optional[dict]:
|
||||
if skip_positions is None:
|
||||
skip_positions = set()
|
||||
|
||||
intent_lower = intent_description.lower()
|
||||
if "first image in explore grid" in intent_lower:
|
||||
grid_items = [
|
||||
n
|
||||
for n in nodes
|
||||
if n.get("y", 9999) < 2000
|
||||
and (
|
||||
"grid card layout container" in (n.get("semantic_string", "") or "").lower()
|
||||
or "image button" in (n.get("semantic_string", "") or "").lower()
|
||||
)
|
||||
and (n.get("x", -1), n.get("y", -1)) not in skip_positions
|
||||
]
|
||||
if grid_items:
|
||||
# Sort by y (row) then by x (col)
|
||||
grid_items.sort(key=lambda n: (n.get("y", 9999), n.get("x", 9999)))
|
||||
return grid_items[0]
|
||||
|
||||
# --- Profile Structural Fast Paths ---
|
||||
if "following list" in intent_lower or "followers list" in intent_lower:
|
||||
target_id = "profile_header_following" if "following" in intent_lower else "profile_header_followers"
|
||||
for n in nodes:
|
||||
res_id = n.get("id", "") or n.get("resource_id", "")
|
||||
if target_id in res_id:
|
||||
return n
|
||||
# Fallback to text matching if ID not found
|
||||
for n in nodes:
|
||||
sem = (n.get("semantic_string", "") or "").lower()
|
||||
desc = (n.get("description", "") or "").lower()
|
||||
text = (n.get("text", "") or "").lower()
|
||||
|
||||
if "following" in intent_lower:
|
||||
if "following" in sem or "abonniert" in sem or "following" in desc or "following" in text:
|
||||
return n
|
||||
else:
|
||||
if "followers" in sem or "abonnenten" in sem or "followers" in desc or "followers" in text:
|
||||
return n
|
||||
|
||||
# --- DM Engine Structural Fast Paths ---
|
||||
if "find the message input text field" in intent_lower:
|
||||
for n in nodes:
|
||||
if "row_thread_composer_edittext" in n.get("id", "") or "row_thread_composer_edittext" in n.get("resource_id", ""):
|
||||
return n
|
||||
|
||||
if "find the send message button" in intent_lower:
|
||||
for n in nodes:
|
||||
if "row_thread_composer_button_send" in n.get("id", "") or "row_thread_composer_button_send" in n.get("resource_id", ""):
|
||||
return n
|
||||
|
||||
if "find unread message threads" in intent_lower:
|
||||
# We must be extremely strict here: It's only unread if it has the "unread" text or indicator dot
|
||||
unread_candidates = []
|
||||
|
||||
# 1. Find all explicit unread dots in the UI
|
||||
dot_nodes = [
|
||||
d for d in nodes
|
||||
if "thread_indicator_status_dot" in (d.get("id", "") or d.get("resource_id", ""))
|
||||
]
|
||||
|
||||
import re
|
||||
|
||||
for n in nodes:
|
||||
is_unread = False
|
||||
res_id = n.get("id", "") or n.get("resource_id", "")
|
||||
|
||||
if "row_inbox_container" in res_id and (n.get("x", -1), n.get("y", -1)) not in skip_positions:
|
||||
content_desc = (n.get("description", "") or "").lower()
|
||||
semantic = (n.get("semantic_string", "") or "").lower()
|
||||
|
||||
# 1. Check for explicit 'unread' in description
|
||||
if "unread" in content_desc or "unread" in semantic:
|
||||
is_unread = True
|
||||
|
||||
# 2. Check if an unread dot falls inside this container's bounds
|
||||
if not is_unread and dot_nodes:
|
||||
bounds_str = n.get("bounds", "")
|
||||
m = re.match(r"\[\d+,(\d+)\]\[\d+,(\d+)\]", bounds_str)
|
||||
if m:
|
||||
y1, y2 = int(m.group(1)), int(m.group(2))
|
||||
for dot in dot_nodes:
|
||||
dot_y = dot.get("y", -1)
|
||||
if y1 <= dot_y <= y2:
|
||||
is_unread = True
|
||||
break
|
||||
|
||||
if is_unread and n.get("y", 0) > 200:
|
||||
unread_candidates.append(n)
|
||||
|
||||
if unread_candidates:
|
||||
unread_candidates.sort(key=lambda n: n.get("y", 9999))
|
||||
return unread_candidates[0]
|
||||
|
||||
if "find the last received message text" in intent_lower:
|
||||
msg_candidates = []
|
||||
for n in nodes:
|
||||
res_id = n.get("id", "") or n.get("resource_id", "")
|
||||
# The actual message text bubble
|
||||
if "direct_text_message_text_view" in res_id or "message_content" in res_id:
|
||||
msg_candidates.append(n)
|
||||
|
||||
if msg_candidates:
|
||||
# Sort by y descending (bottom-most message is the last one)
|
||||
msg_candidates.sort(key=lambda n: n.get("y", 0), reverse=True)
|
||||
return msg_candidates[0]
|
||||
|
||||
return None
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Action Memory Delegation
|
||||
# ──────────────────────────────────────────────
|
||||
@@ -317,36 +228,17 @@ class TelepathicEngine:
|
||||
y = node.get("y", 0)
|
||||
semantic = (node.get("semantic_string", "") or "").lower()
|
||||
|
||||
# 1. Navigation Tab Guard (Must be at the bottom)
|
||||
nav_intents = [
|
||||
"tap direct message icon inbox",
|
||||
"tap inbox",
|
||||
"tap heart icon notifications",
|
||||
"tap home tab",
|
||||
"tap explore tab",
|
||||
"tap reels tab",
|
||||
"tap profile tab",
|
||||
"tap messages tab",
|
||||
]
|
||||
is_nav_intent = any(n in intent for n in nav_intents)
|
||||
if is_nav_intent:
|
||||
if y < screen_height * 0.85:
|
||||
return False
|
||||
return True
|
||||
|
||||
# 2. Block non-nav intents from clicking in the nav zone
|
||||
if y >= screen_height * 0.85:
|
||||
# Not a nav intent, but trying to click the nav bar
|
||||
return False
|
||||
|
||||
# 3. Post Username Guard
|
||||
if "post username" in intent:
|
||||
if "story" in semantic and y < screen_height * 0.2:
|
||||
# 1. Post Username Guard
|
||||
if "post username" in intent or "author username" in intent:
|
||||
if "story" in semantic:
|
||||
# E.g. "Your Story" circle at the top
|
||||
return False
|
||||
# Prevent tapping a search list item when looking for a post username
|
||||
if "row search user container" in semantic.replace("_", " "):
|
||||
return False
|
||||
# Prevent tapping bottom tabs
|
||||
if "tab" in semantic and "exclude bottom tabs" in intent:
|
||||
return False
|
||||
return True
|
||||
|
||||
# 3.5 Media Content Guard
|
||||
@@ -354,6 +246,9 @@ class TelepathicEngine:
|
||||
# Prevent tapping a search keyword instead of a media post
|
||||
if "row search keyword title" in semantic.replace("_", " "):
|
||||
return False
|
||||
# Prevent tapping bottom tabs
|
||||
if "tab" in semantic and "exclude bottom tabs" in intent:
|
||||
return False
|
||||
|
||||
# 3.6 Post Author Username Header Guard
|
||||
if "post author username header" in intent:
|
||||
|
||||
@@ -65,26 +65,29 @@ def _run_zero_latency_unfollow_loop(
|
||||
try:
|
||||
xml_dump = device.dump_hierarchy()
|
||||
|
||||
import re
|
||||
# ── Perimeter Guard: Verify we're still inside Instagram ──
|
||||
if xml_dump:
|
||||
import re
|
||||
|
||||
# Smart Unfollow Phase 1: Find user rows via structural UI markers, not LLM (too prone to hallucinate headers)
|
||||
unfollow_packages = set(re.findall(r'package="([^"]+)"', xml_dump))
|
||||
unfollow_app_id = getattr(device, "app_id", "com.instagram.android")
|
||||
if unfollow_packages and unfollow_app_id not in unfollow_packages:
|
||||
logger.error(
|
||||
f"🚨 [UnfollowLoop] FOREIGN APP DETECTED! Packages: {unfollow_packages}. Aborting loop."
|
||||
)
|
||||
device.press("back")
|
||||
random_sleep(1.0, 1.5)
|
||||
return "CONTEXT_LOST"
|
||||
|
||||
# Autonomously identify user rows via Semantic Extraction
|
||||
telepathic = cognitive_stack.get("telepathic")
|
||||
nodes = []
|
||||
# Find all nodes with resource-id="com.instagram.android:id/follow_list_username"
|
||||
for match in re.finditer(
|
||||
r'resource-id="com\.instagram\.android:id/follow_list_username".*?bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"',
|
||||
xml_dump,
|
||||
):
|
||||
x1, y1, x2, y2 = map(int, match.groups())
|
||||
nodes.append({"x": (x1 + x2) // 2, "y": (y1 + y2) // 2, "bounds": True})
|
||||
|
||||
# Also try com.instagram.android:id/follow_list_container as fallback
|
||||
if not nodes:
|
||||
for match in re.finditer(
|
||||
r'resource-id="com\.instagram\.android:id/follow_list_container".*?bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"',
|
||||
xml_dump,
|
||||
):
|
||||
x1, y1, x2, y2 = map(int, match.groups())
|
||||
nodes.append({"x": (x1 + x2) // 2, "y": (y1 + y2) // 2, "bounds": True})
|
||||
if telepathic:
|
||||
nodes = telepathic._extract_semantic_nodes(
|
||||
xml_dump, "List item containing a user profile image, username, and following/following button"
|
||||
)
|
||||
else:
|
||||
logger.warning("No telepathic engine found, skipping semantic extraction.")
|
||||
|
||||
action_taken = False
|
||||
for node in nodes:
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
import logging
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
from time import sleep
|
||||
|
||||
@@ -95,6 +97,62 @@ def get_value(count, name, default=0):
|
||||
return default
|
||||
|
||||
|
||||
_LEARNED_AD_MARKERS_FILE = os.path.join(os.getcwd(), "learned_ad_markers.json")
|
||||
_LEARNED_AD_MARKERS_CACHE = None
|
||||
|
||||
def get_learned_ad_markers() -> set:
|
||||
global _LEARNED_AD_MARKERS_CACHE
|
||||
if _LEARNED_AD_MARKERS_CACHE is not None:
|
||||
return _LEARNED_AD_MARKERS_CACHE
|
||||
|
||||
if os.path.exists(_LEARNED_AD_MARKERS_FILE):
|
||||
try:
|
||||
with open(_LEARNED_AD_MARKERS_FILE, "r") as f:
|
||||
_LEARNED_AD_MARKERS_CACHE = set(json.load(f))
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to load learned ad markers: {e}")
|
||||
_LEARNED_AD_MARKERS_CACHE = set()
|
||||
else:
|
||||
_LEARNED_AD_MARKERS_CACHE = set()
|
||||
|
||||
return _LEARNED_AD_MARKERS_CACHE
|
||||
|
||||
def learn_ad_marker(marker: str, xml_hierarchy: str):
|
||||
global _LEARNED_AD_MARKERS_CACHE
|
||||
if not marker or len(marker) > 30:
|
||||
return
|
||||
|
||||
marker = marker.strip().lower()
|
||||
|
||||
# Structural verification: the VLM-suggested marker MUST exist as an exact node text/desc in the current UI!
|
||||
import xml.etree.ElementTree as ET
|
||||
try:
|
||||
root = ET.fromstring(xml_hierarchy)
|
||||
found_in_ui = False
|
||||
for node in root.iter("node"):
|
||||
text = node.attrib.get("text", "").strip().lower()
|
||||
desc = node.attrib.get("content-desc", "").strip().lower()
|
||||
if text == marker or desc == marker:
|
||||
found_in_ui = True
|
||||
break
|
||||
|
||||
if not found_in_ui:
|
||||
logger.debug(f"🧠 [Autonomous FSD] Rejected hallucinated Ad marker '{marker}' (not found as exact node match in UI).")
|
||||
return
|
||||
except Exception:
|
||||
return
|
||||
|
||||
markers = get_learned_ad_markers()
|
||||
if marker not in markers and marker not in {"ad", "sponsored", "advertisement", "gesponsert", "anzeige", "werbung"}:
|
||||
markers.add(marker)
|
||||
logger.info(f"🧠 [Autonomous FSD] Verified and Learned new Ad marker: '{marker}'. Persisting for zero-latency detection.", extra={"color": f"{Style.BRIGHT}{Fore.GREEN}"})
|
||||
try:
|
||||
with open(_LEARNED_AD_MARKERS_FILE, "w") as f:
|
||||
json.dump(list(markers), f)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to save learned ad markers: {e}")
|
||||
|
||||
|
||||
def is_ad(xml_hierarchy: str, cognitive_stack: dict = None) -> bool:
|
||||
"""
|
||||
Checks if the current view contains an advertisement using autonomous learning.
|
||||
@@ -125,26 +183,34 @@ def is_ad(xml_hierarchy: str, cognitive_stack: dict = None) -> bool:
|
||||
# Standalone label patterns: match only when the text/desc IS the ad marker,
|
||||
# not when "ad" appears inside longer phrases like "Create messaging ad"
|
||||
AD_EXACT_LABELS = {"ad", "sponsored", "advertisement", "gesponsert", "anzeige", "werbung"}
|
||||
AD_EXACT_LABELS.update(get_learned_ad_markers())
|
||||
|
||||
try:
|
||||
root = ET.fromstring(xml_hierarchy)
|
||||
|
||||
# Check if we are in a feed (to prevent false positives on profiles with 'Ad Tools' buttons)
|
||||
from GramAddict.core.perception.feed_analysis import FEED_MARKERS
|
||||
in_feed = any(marker in xml_hierarchy for marker in FEED_MARKERS)
|
||||
|
||||
for node in root.iter("node"):
|
||||
attrib = node.attrib
|
||||
content_desc = attrib.get("content-desc", "")
|
||||
text = attrib.get("text", "")
|
||||
res_id = attrib.get("resource-id", "")
|
||||
|
||||
# Structural check (Instagram specific)
|
||||
# Structural check (Instagram specific) is always trusted
|
||||
if any(marker_id in res_id for marker_id in AD_RESOURCE_IDS):
|
||||
return True
|
||||
|
||||
# Exact label match: only trigger when the entire text/desc
|
||||
# IS an ad marker (e.g. text="Ad", content-desc="Sponsored")
|
||||
# This prevents false positives from "Create messaging ad"
|
||||
if text.strip().lower() in AD_EXACT_LABELS:
|
||||
return True
|
||||
if content_desc.strip().lower() in AD_EXACT_LABELS:
|
||||
return True
|
||||
# We ONLY trust this if we are actually in a feed, to prevent triggering
|
||||
# on the "Ad Tools" / "Ad" buttons present on business profiles.
|
||||
if in_feed:
|
||||
if text.strip().lower() in AD_EXACT_LABELS:
|
||||
return True
|
||||
if content_desc.strip().lower() in AD_EXACT_LABELS:
|
||||
return True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
@@ -21,6 +21,7 @@ If Instagram updates its app and moves a button, GramPilot doesn't crash. It fal
|
||||
## ✨ Core Features
|
||||
|
||||
* 🚫 **Zero Limits Configuration**: Forget about configuring "max_likes" or "delays". GramPilot uses a **Dopamine Pacing Engine** to simulate human boredom. If the content isn't interesting, it skips it or ends the session early.
|
||||
* 🎯 **Mission-Driven Navigation**: Say goodbye to abstract goal configurations. Define a `strategy` (like `aggressive_growth` or `nurture_community`) in `config.yml`, and the **Goal Decomposer Engine** automatically orchestrates the optimal routing and task allocation using enabled plugins.
|
||||
* ⚖️ **Active Inference (Shadow Mode)**: The bot continuously predicts the outcome of its clicks. If it lands on a popup instead of a profile, it registers a "Prediction Error", presses back, and dynamically recalibrates without panicking.
|
||||
* ⛩️ **Telepathic Engine**: A strictly tiered resolution cascade (Keyword -> Vectors -> LLM) that ensures 90% of navigation happens at 0-token cost while maintaining fallback AI resilience.
|
||||
* 🧬 **Resonance Oracle**: The bot only interacts with content that matches a pre-defined persona aesthetic, completely bypassing spam or low-quality content.
|
||||
|
||||
@@ -9,11 +9,14 @@ from datetime import datetime
|
||||
# Add root project path so we can import internal modules safely
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from GramAddict.core.llm_provider import query_telepathic_llm
|
||||
from GramAddict.core.llm_provider import query_llm, query_telepathic_llm
|
||||
|
||||
BENCHMARKS_FILE = os.path.join(os.path.dirname(__file__), "data/llm_benchmarks.json")
|
||||
SCENARIOS_FILE = os.path.join(os.path.dirname(__file__), "data/benchmark_scenarios.json")
|
||||
|
||||
# Minimum iterations for statistical significance
|
||||
MIN_ITERATIONS = 5
|
||||
|
||||
|
||||
def load_json(path):
|
||||
if os.path.exists(path):
|
||||
@@ -31,35 +34,37 @@ def save_json(path, data):
|
||||
|
||||
|
||||
def normalize_scores(db):
|
||||
"""Normalize relative performance by AVERAGE score per scenario, not raw totals."""
|
||||
if not db.get("models"):
|
||||
return db
|
||||
|
||||
# 1. Find the highest raw score across all models
|
||||
max_raw = 0
|
||||
max_avg = 0
|
||||
leader_model = None
|
||||
|
||||
for name, data in db["models"].items():
|
||||
if data.get("is_unsuitable"):
|
||||
continue
|
||||
|
||||
raw = data.get("raw_score", 0)
|
||||
if raw > max_raw:
|
||||
max_raw = raw
|
||||
scenario_count = data.get("scenario_count", 1)
|
||||
avg = data.get("raw_score", 0) / max(scenario_count, 1)
|
||||
data["avg_score_per_scenario"] = round(avg, 1)
|
||||
|
||||
if avg > max_avg:
|
||||
max_avg = avg
|
||||
leader_model = name
|
||||
elif raw == max_raw and max_raw > 0:
|
||||
# Tie-breaker: Latency
|
||||
elif avg == max_avg and max_avg > 0:
|
||||
current_lat = data.get("latency_ms", 99999)
|
||||
leader_lat = db["models"][leader_model].get("latency_ms", 99999)
|
||||
if current_lat < leader_lat:
|
||||
leader_model = name
|
||||
|
||||
if max_raw == 0:
|
||||
if max_avg == 0:
|
||||
return db
|
||||
|
||||
# 2. Update relative performance
|
||||
for name, data in db["models"].items():
|
||||
raw = data.get("raw_score", 0)
|
||||
data["relative_performance_pct"] = round((raw / max_raw) * 100, 1)
|
||||
scenario_count = data.get("scenario_count", 1)
|
||||
avg = data.get("raw_score", 0) / max(scenario_count, 1)
|
||||
data["relative_performance_pct"] = round((avg / max_avg) * 100, 1)
|
||||
data["is_leader"] = name == leader_model
|
||||
|
||||
return db
|
||||
@@ -75,21 +80,15 @@ def get_installed_ollama_models():
|
||||
models = []
|
||||
for line in output.split("\n")[1:]:
|
||||
if line.strip():
|
||||
# Format: NAME, ID, SIZE, MODIFIED
|
||||
parts = line.split()
|
||||
if len(parts) >= 3:
|
||||
name = parts[0]
|
||||
size = parts[2]
|
||||
|
||||
# 1. Skip if size is '-' (remote/cloud model)
|
||||
if size == "-":
|
||||
continue
|
||||
|
||||
# 2. Skip ':cloud' tagged models explicitly
|
||||
if ":cloud" in name:
|
||||
continue
|
||||
|
||||
# 3. Filter out purely embedding models
|
||||
if any(k in name.lower() for k in ["embed", "minilm", "rerank"]):
|
||||
continue
|
||||
|
||||
@@ -100,7 +99,131 @@ def get_installed_ollama_models():
|
||||
return []
|
||||
|
||||
|
||||
def benchmark_model(model_name: str, url: str, force: bool = False, iterations: int = 3):
|
||||
def _run_telepathic_scenario(scenario, model_name, url, iterations):
|
||||
"""Run a telepathic (JSON element selection) scenario."""
|
||||
system_prompt = (
|
||||
"You identify which UI element to tap based ONLY on a JSON array of parsed Android elements. "
|
||||
'Output ONLY valid JSON: {"index": number, "reason": "brief reason"}'
|
||||
)
|
||||
|
||||
user_prompt = (
|
||||
f"Which element should I tap to: {scenario['task']}\n\n"
|
||||
f"Elements:\n{json.dumps(scenario['nodes'], indent=1)}\n\n"
|
||||
"Rules:\n"
|
||||
"- Pick the SMALLEST, most specific button or icon\n"
|
||||
"- NEVER pick large containers\n"
|
||||
'Return: {"index": number, "reason": "..."}'
|
||||
)
|
||||
|
||||
latencies = []
|
||||
scores = []
|
||||
successes = 0
|
||||
|
||||
for _ in range(iterations):
|
||||
start_time = time.time()
|
||||
try:
|
||||
resp_str = query_telepathic_llm(model_name, url, system_prompt, user_prompt)
|
||||
latency = int((time.time() - start_time) * 1000)
|
||||
latencies.append(latency)
|
||||
except Exception as e:
|
||||
print(f" ❌ API Request failed: {e}")
|
||||
scores.append(0)
|
||||
continue
|
||||
|
||||
raw_points = 0
|
||||
try:
|
||||
clean = resp_str.strip()
|
||||
if clean.startswith("```json"):
|
||||
clean = clean[7:]
|
||||
if clean.endswith("```"):
|
||||
clean = clean[:-3]
|
||||
data = json.loads(clean)
|
||||
|
||||
if "index" in data and "reason" in data:
|
||||
raw_points += 40
|
||||
if data["index"] == scenario["target_index"]:
|
||||
raw_points += 60
|
||||
successes += 1
|
||||
else:
|
||||
print(f" ❌ Wrong index ({data.get('index')}). Target was {scenario['target_index']}.")
|
||||
else:
|
||||
print(" ❌ JSON missing fields.")
|
||||
except Exception:
|
||||
print(" ❌ JSON Parsing failed.")
|
||||
|
||||
scores.append(raw_points)
|
||||
|
||||
return scores, latencies, successes
|
||||
|
||||
|
||||
def _run_brain_scenario(scenario, model_name, url, iterations):
|
||||
"""Run a brain action extraction scenario (format_json=False)."""
|
||||
system_prompt = (
|
||||
f"You are an autonomous Instagram agent. Your goal is: '{scenario['task']}'.\n"
|
||||
f"You are currently on screen: {scenario['screen_type']}.\n"
|
||||
f"Available actions: {scenario['available_actions']}\n"
|
||||
"INSTRUCTIONS: Reply with ONLY the action string. Nothing else."
|
||||
)
|
||||
|
||||
user_prompt = "Choose the next best action."
|
||||
|
||||
latencies = []
|
||||
scores = []
|
||||
successes = 0
|
||||
|
||||
for _ in range(iterations):
|
||||
start_time = time.time()
|
||||
try:
|
||||
# CRITICAL: Use format_json=False — this is the Brain code path
|
||||
ans = query_llm(
|
||||
url=url,
|
||||
model=model_name,
|
||||
prompt=user_prompt,
|
||||
system=system_prompt,
|
||||
format_json=False,
|
||||
timeout=30,
|
||||
temperature=0.0,
|
||||
max_tokens=50,
|
||||
)
|
||||
latency = int((time.time() - start_time) * 1000)
|
||||
latencies.append(latency)
|
||||
except Exception as e:
|
||||
print(f" ❌ API Request failed: {e}")
|
||||
scores.append(0)
|
||||
continue
|
||||
|
||||
raw_points = 0
|
||||
if ans and "response" in ans:
|
||||
response = ans["response"].strip().lower()
|
||||
|
||||
# Points for structural adherence (returned a clean string)
|
||||
if response and response in [a.lower() for a in scenario["available_actions"]]:
|
||||
raw_points += 40
|
||||
|
||||
# Points for correctness
|
||||
if scenario.get("accept_any_valid"):
|
||||
# Any valid action from the list is acceptable
|
||||
raw_points += 60
|
||||
successes += 1
|
||||
elif response == scenario["target_action"].lower():
|
||||
raw_points += 60
|
||||
successes += 1
|
||||
else:
|
||||
print(f" ⚠️ Valid but suboptimal: '{response}' (target: '{scenario['target_action']}')")
|
||||
raw_points += 20 # Partial credit for valid but wrong action
|
||||
else:
|
||||
print(f" ❌ Invalid response: '{response}' not in available actions")
|
||||
else:
|
||||
print(" ❌ Empty or null response from LLM")
|
||||
|
||||
scores.append(raw_points)
|
||||
|
||||
return scores, latencies, successes
|
||||
|
||||
|
||||
def benchmark_model(model_name: str, url: str, force: bool = False, iterations: int = MIN_ITERATIONS):
|
||||
iterations = max(iterations, MIN_ITERATIONS) # Enforce minimum
|
||||
|
||||
db = load_json(BENCHMARKS_FILE) or {"models": {}}
|
||||
scenarios_data = load_json(SCENARIOS_FILE)
|
||||
if not scenarios_data:
|
||||
@@ -113,95 +236,46 @@ def benchmark_model(model_name: str, url: str, force: bool = False, iterations:
|
||||
print(f"Typical execution skip for {model_name} (Rel: {pct}%). Use --force.")
|
||||
return
|
||||
|
||||
print(f"\n🚀 [Competitive Benchmarking] Model: {model_name}")
|
||||
print(f"\n🚀 [Competitive Benchmarking] Model: {model_name} ({iterations} iterations)")
|
||||
|
||||
total_raw = 0
|
||||
total_latency = 0
|
||||
results_detail = {}
|
||||
passed_all = True
|
||||
|
||||
system_prompt = (
|
||||
"You identify which UI element to tap based ONLY on a JSON array of parsed Android elements. "
|
||||
'Output ONLY valid JSON: {"index": number, "reason": "brief reason"}'
|
||||
)
|
||||
|
||||
scenarios = scenarios_data["scenarios"]
|
||||
for scenario in scenarios:
|
||||
print(f"--- Running: {scenario['name']} ---")
|
||||
scenario_type = scenario.get("type", "telepathic")
|
||||
print(f"--- [{scenario_type.upper()}] {scenario['name']} ---")
|
||||
|
||||
user_prompt = (
|
||||
f"Which element should I tap to: {scenario['task']}\n\n"
|
||||
f"Elements:\n{json.dumps(scenario['nodes'], indent=1)}\n\n"
|
||||
"Rules:\n"
|
||||
"- Pick the SMALLEST, most specific button or icon\n"
|
||||
"- NEVER pick large containers\n"
|
||||
"Return: {\"index\": number, \"reason\": \"...\"}"
|
||||
)
|
||||
|
||||
scenario_latencies = []
|
||||
scenario_scores = []
|
||||
successes = 0
|
||||
|
||||
for _ in range(iterations):
|
||||
start_time = time.time()
|
||||
try:
|
||||
resp_str = query_telepathic_llm(model_name, url, system_prompt, user_prompt)
|
||||
latency = int((time.time() - start_time) * 1000)
|
||||
scenario_latencies.append(latency)
|
||||
except Exception as e:
|
||||
print(f" ❌ API Request failed for scenario {scenario['id']}: {e}")
|
||||
passed_all = False
|
||||
continue
|
||||
|
||||
raw_points = 0
|
||||
try:
|
||||
clean = resp_str.strip()
|
||||
if clean.startswith("```json"):
|
||||
clean = clean[7:]
|
||||
if clean.endswith("```"):
|
||||
clean = clean[:-3]
|
||||
data = json.loads(clean)
|
||||
|
||||
# Points for structural adherence
|
||||
if "index" in data and "reason" in data:
|
||||
raw_points += 40
|
||||
|
||||
# Points for correctness
|
||||
if data["index"] == scenario["target_index"]:
|
||||
raw_points += 60
|
||||
successes += 1
|
||||
else:
|
||||
print(f" ❌ Wrong index ({data.get('index')}). Target was {scenario['target_index']}.")
|
||||
else:
|
||||
print(" ❌ JSON missing fields.")
|
||||
except Exception:
|
||||
print(" ❌ JSON Parsing failed.")
|
||||
|
||||
scenario_scores.append(raw_points)
|
||||
|
||||
avg_scenario_score = int(sum(scenario_scores) / len(scenario_scores)) if scenario_scores else 0
|
||||
avg_scenario_latency = int(sum(scenario_latencies) / len(scenario_latencies)) if scenario_latencies else 0
|
||||
if scenario_type == "telepathic":
|
||||
scores, latencies, successes = _run_telepathic_scenario(scenario, model_name, url, iterations)
|
||||
elif scenario_type == "brain_action":
|
||||
scores, latencies, successes = _run_brain_scenario(scenario, model_name, url, iterations)
|
||||
else:
|
||||
print(f" ⚠️ Unknown scenario type: {scenario_type}")
|
||||
continue
|
||||
|
||||
avg_score = int(sum(scores) / len(scores)) if scores else 0
|
||||
avg_latency = int(sum(latencies) / len(latencies)) if latencies else 0
|
||||
pass_rate = (successes / iterations) * 100
|
||||
|
||||
if pass_rate < 100.0:
|
||||
passed_all = False
|
||||
|
||||
print(
|
||||
f" Result: {pass_rate:.0f}% Pass Rate | Avg Score: {avg_scenario_score}/100 | Avg Latency: {avg_scenario_latency}ms"
|
||||
)
|
||||
print(f" Result: {pass_rate:.0f}% Pass | Avg Score: {avg_score}/100 | Avg Latency: {avg_latency}ms")
|
||||
|
||||
# Consistent format: always an object
|
||||
results_detail[scenario["id"]] = {
|
||||
"avg_score": avg_scenario_score,
|
||||
"avg_score": avg_score,
|
||||
"pass_rate": pass_rate,
|
||||
"latency": avg_scenario_latency,
|
||||
"latency": avg_latency,
|
||||
}
|
||||
total_raw += avg_scenario_score
|
||||
total_latency += avg_scenario_latency
|
||||
total_raw += avg_score
|
||||
total_latency += avg_latency
|
||||
|
||||
avg_latency = total_latency // len(scenarios) if scenarios else 0
|
||||
print(
|
||||
f"\n📊 {model_name} Result: {'PASS' if passed_all else 'FAIL'} | Avg Score: {total_raw} | Latency: {avg_latency}ms"
|
||||
)
|
||||
print(f"\n📊 {model_name}: {'PASS' if passed_all else 'FAIL'} | Total: {total_raw} | Latency: {avg_latency}ms")
|
||||
|
||||
if model_name not in db["models"]:
|
||||
db["models"][model_name] = {}
|
||||
@@ -209,16 +283,17 @@ def benchmark_model(model_name: str, url: str, force: bool = False, iterations:
|
||||
db["models"][model_name].update(
|
||||
{
|
||||
"raw_score": total_raw,
|
||||
"scenario_count": len(scenarios),
|
||||
"telepathic_score": int((total_raw / (len(scenarios) * 100)) * 100) if scenarios else 0,
|
||||
"latency_ms": avg_latency,
|
||||
"last_tested": datetime.utcnow().isoformat() + "Z",
|
||||
"details": results_detail,
|
||||
"passed_all": passed_all,
|
||||
"is_unsuitable": not passed_all,
|
||||
"iterations": iterations,
|
||||
}
|
||||
)
|
||||
|
||||
# Recalculate relative scores across all models
|
||||
db = normalize_scores(db)
|
||||
save_json(BENCHMARKS_FILE, db)
|
||||
|
||||
@@ -233,7 +308,7 @@ if __name__ == "__main__":
|
||||
parser.add_argument("--force", action="store_true", help="Force re-testing")
|
||||
parser.add_argument("--all-ollama", action="store_true", help="Automatically find and test all local Ollama models")
|
||||
parser.add_argument(
|
||||
"--iterations", type=int, default=3, help="Number of iterations per scenario to measure reliability"
|
||||
"--iterations", type=int, default=MIN_ITERATIONS, help=f"Iterations per scenario (min: {MIN_ITERATIONS})"
|
||||
)
|
||||
|
||||
args, unknown = parser.parse_known_args()
|
||||
|
||||
@@ -7,7 +7,7 @@ emoji==2.12.1
|
||||
langdetect==1.0.9
|
||||
atomicwrites==1.4.1
|
||||
spintax==1.0.4
|
||||
requests>=2.31.0
|
||||
requests>=2.32.0
|
||||
packaging>=23.0
|
||||
python-dotenv==1.0.1
|
||||
qdrant-client>=1.7.0
|
||||
|
||||
@@ -26,10 +26,26 @@ else
|
||||
elif [ -f "$core_test_file" ]; then
|
||||
TEST_TARGETS="$TEST_TARGETS $core_test_file"
|
||||
else
|
||||
# If no direct unit test, fallback to running all unit tests to be safe
|
||||
echo "⚠️ No direct unit test found for $file, falling back to all unit tests."
|
||||
TEST_TARGETS="tests/unit"
|
||||
break
|
||||
# Try to find matching e2e tests by searching for each word in the module name
|
||||
module_name="${filename%.py}"
|
||||
e2e_matches=""
|
||||
for word in $(echo "$module_name" | tr '_' '\n'); do
|
||||
if [ ${#word} -ge 4 ]; then # Only search meaningful words (4+ chars)
|
||||
found=$(find tests/e2e -name "test_*${word}*.py" 2>/dev/null | head -3)
|
||||
if [ -n "$found" ]; then
|
||||
e2e_matches="$e2e_matches $found"
|
||||
fi
|
||||
fi
|
||||
done
|
||||
e2e_matches=$(echo "$e2e_matches" | xargs -n1 2>/dev/null | sort -u | head -3 | xargs 2>/dev/null)
|
||||
if [ -n "$e2e_matches" ]; then
|
||||
echo "⚠️ No direct unit test for $file, using matching E2E tests: $e2e_matches"
|
||||
TEST_TARGETS="$TEST_TARGETS $e2e_matches"
|
||||
else
|
||||
echo "⚠️ No direct unit test found for $file, falling back to all unit tests."
|
||||
TEST_TARGETS="tests/unit"
|
||||
break
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
done
|
||||
|
||||
BIN
tests/.DS_Store
vendored
Normal file
BIN
tests/.DS_Store
vendored
Normal file
Binary file not shown.
BIN
tests/__pycache__/__init__.cpython-311.pyc
Normal file
BIN
tests/__pycache__/__init__.cpython-311.pyc
Normal file
Binary file not shown.
BIN
tests/__pycache__/conftest.cpython-311-pytest-8.3.5.pyc
Normal file
BIN
tests/__pycache__/conftest.cpython-311-pytest-8.3.5.pyc
Normal file
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
tests/anomalies/__pycache__/__init__.cpython-311.pyc
Normal file
BIN
tests/anomalies/__pycache__/__init__.cpython-311.pyc
Normal file
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
tests/chaos/__pycache__/__init__.cpython-311.pyc
Normal file
BIN
tests/chaos/__pycache__/__init__.cpython-311.pyc
Normal file
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -36,3 +36,67 @@ def _isolate_config_from_argparse(monkeypatch):
|
||||
|
||||
def pytest_configure(config):
|
||||
config.addinivalue_line("markers", "live_llm: requires a running local LLM (Ollama)")
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════
|
||||
# PERMANENT MOCK BAN — Zero-Tolerance Enforcement
|
||||
# ═══════════════════════════════════════════════════════
|
||||
|
||||
_BANNED_PATTERNS = (
|
||||
"from unittest.mock",
|
||||
"from unittest import mock",
|
||||
"import unittest.mock",
|
||||
"from mock import",
|
||||
"import mock",
|
||||
"MagicMock(",
|
||||
"MagicMock)",
|
||||
"@patch(",
|
||||
"@patch\n",
|
||||
"patch.object(",
|
||||
)
|
||||
|
||||
|
||||
def pytest_collect_file(parent, file_path):
|
||||
"""Scan every collected .py test file for banned mock imports.
|
||||
|
||||
This runs at COLLECTION TIME — before any test executes.
|
||||
If a banned pattern is found, the file is still collected but
|
||||
every test inside it will be marked as an error via
|
||||
pytest_collection_modifyitems below.
|
||||
"""
|
||||
if file_path.suffix == ".py" and file_path.name.startswith("test_"):
|
||||
try:
|
||||
content = file_path.read_text(encoding="utf-8")
|
||||
for pattern in _BANNED_PATTERNS:
|
||||
if pattern in content:
|
||||
# Store the violation on the config for later reporting
|
||||
if not hasattr(parent.config, "_mock_violations"):
|
||||
parent.config._mock_violations = {}
|
||||
parent.config._mock_violations[str(file_path)] = pattern
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
return None # Let pytest's default collector handle the file
|
||||
|
||||
|
||||
def pytest_collection_modifyitems(config, items):
|
||||
"""Fail every test from a file that contains banned mock patterns."""
|
||||
violations = getattr(config, "_mock_violations", {})
|
||||
if not violations:
|
||||
return
|
||||
|
||||
for item in items:
|
||||
test_file = str(item.fspath)
|
||||
if test_file in violations:
|
||||
pattern = violations[test_file]
|
||||
item.add_marker(
|
||||
pytest.mark.xfail(
|
||||
reason=(
|
||||
f"🚨 MOCK BAN VIOLATION: File contains '{pattern}'. "
|
||||
f"unittest.mock is permanently banned. "
|
||||
f"Use monkeypatch + real fixtures instead."
|
||||
),
|
||||
strict=True,
|
||||
raises=Exception,
|
||||
)
|
||||
)
|
||||
|
||||
Binary file not shown.
BIN
tests/core/__pycache__/test_config.cpython-311-pytest-8.3.5.pyc
Normal file
BIN
tests/core/__pycache__/test_config.cpython-311-pytest-8.3.5.pyc
Normal file
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
56
tests/core/test_action_memory_fast_path.py
Normal file
56
tests/core/test_action_memory_fast_path.py
Normal file
@@ -0,0 +1,56 @@
|
||||
from GramAddict.core.perception.action_memory import ActionMemory
|
||||
from GramAddict.core.perception.screen_identity import ScreenIdentity, ScreenType
|
||||
|
||||
|
||||
def test_screen_identity_detects_other_profile_with_missing_header_container():
|
||||
"""
|
||||
RED: ScreenIdentity used to misclassify OTHER_PROFILE as UNKNOWN or POST_DETAIL
|
||||
if `profile_header_container` was missing, even though `row_profile_header_imageview`
|
||||
was present.
|
||||
GREEN: We added row_profile_header_imageview and profile_tabs_container.
|
||||
"""
|
||||
identity = ScreenIdentity(bot_username="my_bot")
|
||||
|
||||
# Simulate a profile screen that is missing the main container but has the imageview
|
||||
xml_dump = """<?xml version='1.0' encoding='UTF-8' standalone='yes' ?>
|
||||
<hierarchy>
|
||||
<node package="com.instagram.android" resource-id="com.instagram.android:id/row_profile_header_imageview" bounds="[0,0][100,100]" clickable="true"/>
|
||||
<node package="com.instagram.android" resource-id="com.instagram.android:id/profile_tabs_container" bounds="[0,200][1080,300]"/>
|
||||
<node package="com.instagram.android" resource-id="com.instagram.android:id/action_bar_title" text="justkay"/>
|
||||
</hierarchy>
|
||||
"""
|
||||
|
||||
result = identity.identify(xml_dump)
|
||||
assert result["screen_type"] == ScreenType.OTHER_PROFILE
|
||||
|
||||
|
||||
def test_action_memory_verifies_profile_navigation_in_o1_without_vlm(monkeypatch):
|
||||
"""
|
||||
RED: ActionMemory.verify_success used to fallback to VLM because navigating to a profile
|
||||
caused a huge delta, but there was no explicit fast-path for 'profile', causing it
|
||||
to hit `confidence < 0.95` and invoke `evaluator._query_vlm`.
|
||||
GREEN: It now structurally verifies `profile_header_container` instantly.
|
||||
"""
|
||||
memory = ActionMemory()
|
||||
|
||||
class DummyDevice:
|
||||
def get_screenshot_b64(self):
|
||||
raise Exception("VLM SHOULD NOT BE CALLED!")
|
||||
|
||||
device = DummyDevice()
|
||||
|
||||
intent = "tap post username"
|
||||
pre_xml = "<hierarchy><node/></hierarchy>"
|
||||
# Post XML contains profile_header_container!
|
||||
post_xml = "<hierarchy><node resource-id='com.instagram.android:id/profile_header_container'/></hierarchy>"
|
||||
|
||||
# If the fast-path works, it will return True instantly and NOT call get_screenshot_b64
|
||||
success = memory.verify_success(
|
||||
intent=intent,
|
||||
pre_click_xml=pre_xml,
|
||||
post_click_xml=post_xml,
|
||||
device=device,
|
||||
confidence=0.0, # low confidence triggers VLM fallback if fast-path is missing
|
||||
)
|
||||
|
||||
assert success is True
|
||||
@@ -30,7 +30,6 @@ def test_parse_args_no_exit_when_config_loaded(monkeypatch):
|
||||
but a config file is loaded, parse_args() should NOT print help and exit.
|
||||
"""
|
||||
import sys
|
||||
from unittest.mock import patch
|
||||
|
||||
# Simulate running without arguments
|
||||
monkeypatch.setattr(sys, "argv", ["run.py"])
|
||||
@@ -40,13 +39,18 @@ def test_parse_args_no_exit_when_config_loaded(monkeypatch):
|
||||
# Simulate that we successfully loaded a config dictionary (e.g. from config.yml)
|
||||
config.config = {"some_setting": "value"}
|
||||
|
||||
help_called = []
|
||||
def mock_print_help(*args, **kwargs):
|
||||
help_called.append(True)
|
||||
|
||||
monkeypatch.setattr(config.parser, "print_help", mock_print_help)
|
||||
|
||||
# If parse_args() calls exit(0), it will raise SystemExit
|
||||
try:
|
||||
with patch.object(config.parser, "print_help") as mock_print_help:
|
||||
config.parse_args()
|
||||
# If we get here, no exit() was called.
|
||||
# Also, print_help should not have been called.
|
||||
mock_print_help.assert_not_called()
|
||||
config.parse_args()
|
||||
# If we get here, no exit() was called.
|
||||
# Also, print_help should not have been called.
|
||||
assert not help_called, "print_help should not have been called"
|
||||
except SystemExit:
|
||||
import pytest
|
||||
|
||||
|
||||
51
tests/core/test_intent_resolver_multilingual.py
Normal file
51
tests/core/test_intent_resolver_multilingual.py
Normal file
@@ -0,0 +1,51 @@
|
||||
from GramAddict.core.perception.intent_resolver import IntentResolver
|
||||
from GramAddict.core.perception.spatial_parser import SpatialNode
|
||||
|
||||
|
||||
def test_semantic_guard_allows_multilingual_follow_button():
|
||||
"""
|
||||
RED: The IntentResolver's Semantic Guard used to hard-filter for EXACT quotes.
|
||||
If the plugin requested "tap 'Follow' button", but the UI was in German ("Abonnieren"),
|
||||
the Semantic Guard would block it, causing the bot to never follow anyone.
|
||||
GREEN: We added multilingual equivalents to the Semantic Guard logic.
|
||||
"""
|
||||
resolver = IntentResolver()
|
||||
|
||||
intent = "tap 'Follow' button"
|
||||
|
||||
# Create a node that represents a German follow button
|
||||
german_node = SpatialNode(
|
||||
resource_id="com.instagram.android:id/profile_header_follow_button",
|
||||
content_desc="Abonnieren",
|
||||
text="Abonnieren",
|
||||
bounds=(0, 0, 100, 100),
|
||||
)
|
||||
|
||||
# Run the resolver without a device (forces semantic/structural resolution, bypasses VLM)
|
||||
result = resolver.resolve(intent, candidates=[german_node], device=None, screen_height=2000)
|
||||
|
||||
assert result is not None
|
||||
assert result.text == "Abonnieren"
|
||||
|
||||
|
||||
def test_semantic_guard_allows_multilingual_following_button():
|
||||
"""
|
||||
Ensures that "tap 'Following' button" resolves correctly for German UIs
|
||||
("Abonniert", "Gefolgt", "Angefragt").
|
||||
"""
|
||||
resolver = IntentResolver()
|
||||
|
||||
intent = "tap 'Following' button"
|
||||
|
||||
# Create a node that represents a German "Following" button
|
||||
german_node = SpatialNode(
|
||||
resource_id="com.instagram.android:id/profile_header_follow_button",
|
||||
content_desc="Abonniert",
|
||||
text="Abonniert",
|
||||
bounds=(0, 0, 100, 100),
|
||||
)
|
||||
|
||||
result = resolver.resolve(intent, candidates=[german_node], device=None, screen_height=2000)
|
||||
|
||||
assert result is not None
|
||||
assert result.text == "Abonniert"
|
||||
@@ -1,24 +0,0 @@
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
|
||||
from GramAddict.core.qdrant_memory import QdrantBase
|
||||
|
||||
|
||||
def test_get_embedding_api_error_crashes_loudly():
|
||||
"""
|
||||
Test that when the embedding API returns a 500 error,
|
||||
_get_embedding does NOT silently swallow it and return None,
|
||||
but instead crashes loud and fast.
|
||||
"""
|
||||
db = QdrantBase(collection_name="test_collection")
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.status_code = 500
|
||||
mock_response.text = '{"error":"the input length exceeds the context length"}'
|
||||
mock_response.raise_for_status.side_effect = requests.exceptions.HTTPError("500 Server Error")
|
||||
|
||||
with patch("requests.post", return_value=mock_response):
|
||||
with pytest.raises(requests.exceptions.HTTPError):
|
||||
db._get_embedding("some very long text")
|
||||
@@ -1,78 +0,0 @@
|
||||
"""
|
||||
Unfollow Engine Integration Tests
|
||||
=================================
|
||||
Tests Unfollow Engine autonomous loop using real XML hierarchy fixtures
|
||||
to ensure it interacts correctly with the UI instead of relying on
|
||||
false-positive mocks.
|
||||
"""
|
||||
|
||||
import os
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from GramAddict.core.unfollow_engine import _run_zero_latency_unfollow_loop
|
||||
|
||||
FIX_DIR = os.path.join(os.path.dirname(os.path.dirname(__file__)), "fixtures")
|
||||
|
||||
|
||||
def _get_fixture(name: str) -> str:
|
||||
with open(os.path.join(FIX_DIR, name), "r", encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def test_unfollow_engine_extracts_users_and_calls_back_on_high_resonance():
|
||||
"""
|
||||
Test: The unfollow engine must accurately extract user rows from a REAL XML dump
|
||||
and tap them. If resonance is high (user should be kept), it must navigate back.
|
||||
"""
|
||||
# Provide the REAL unfollow list dump
|
||||
real_xml = _get_fixture("unfollow_list_dump.xml")
|
||||
|
||||
device = MagicMock()
|
||||
device.get_info.return_value = {"displayWidth": 1080, "displayHeight": 2400}
|
||||
# It will dump the list, then we simulate going back to it
|
||||
device.dump_hierarchy.return_value = real_xml
|
||||
|
||||
zero_engine = MagicMock()
|
||||
nav_graph = MagicMock()
|
||||
configs = MagicMock()
|
||||
configs.args.total_unfollows_limit = 50
|
||||
|
||||
session_state = MagicMock()
|
||||
session_state.check_limit.return_value = False
|
||||
session_state.totalUnfollowed = 0
|
||||
|
||||
telepathic = MagicMock()
|
||||
# In the unfollow loop, it uses structural markers first (re.finditer), NOT telepathic,
|
||||
# so we don't need to mock telepathic._extract_semantic_nodes for the list itself.
|
||||
# We DO need it to return an empty list when looking for the 'Following' button
|
||||
# so that it simulates "button not found" or "kept user" and hits device.back().
|
||||
telepathic._extract_semantic_nodes.return_value = []
|
||||
|
||||
dopamine = MagicMock()
|
||||
# Let the loop run exactly once (it will process the first user, then we end session)
|
||||
dopamine.is_app_session_over.side_effect = [False, True]
|
||||
dopamine.wants_to_change_feed.return_value = False
|
||||
|
||||
resonance = MagicMock()
|
||||
# High resonance = keep following -> should call back()
|
||||
resonance.calculate_resonance.return_value = 0.9
|
||||
|
||||
cognitive_stack = {"telepathic": telepathic, "dopamine": dopamine, "resonance": resonance}
|
||||
|
||||
_run_zero_latency_unfollow_loop(
|
||||
device, zero_engine, nav_graph, configs, session_state, "some_target", cognitive_stack
|
||||
)
|
||||
|
||||
# In the real XML, the first user is me.and.eloise at bounds [247,1014][537,1061].
|
||||
# Center is (392, 1037). Wait, the engine taps the row, let's see if it taps near there.
|
||||
# The exact math in the engine:
|
||||
# x1, y1, x2, y2 = 247, 1014, 537, 1061
|
||||
# x = (247+537)//2 = 392. y = (1014+1061)//2 = 1037.
|
||||
# It calls _humanized_click(device, x, y) which ultimately does device.click(x, y).
|
||||
# BUT _humanized_click uses gaussian distribution so exact coordinates are fuzzy.
|
||||
|
||||
# The critical assertion: we MUST have pressed back to return to the list.
|
||||
assert device.back.call_count >= 1, "Engine failed to press back after inspecting profile!"
|
||||
|
||||
# And we must have attempted a click on the profile
|
||||
assert device.shell.call_count >= 1, "Engine failed to tap the profile row from the real XML!"
|
||||
147
tests/core/test_zero_maintenance_strings.py
Normal file
147
tests/core/test_zero_maintenance_strings.py
Normal file
@@ -0,0 +1,147 @@
|
||||
"""
|
||||
TDD Tests: Zero-Maintenance String Compliance
|
||||
|
||||
These tests enforce that no hardcoded German/localized strings
|
||||
exist in navigation-critical code paths. The bot must rely
|
||||
exclusively on structural resource_id patterns, never on
|
||||
localized UI text that changes with device language.
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════
|
||||
# 1. TOGGLE_INTENT_MARKERS must be language-agnostic
|
||||
# ══════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestToggleIntentMarkersAreLanguageAgnostic:
|
||||
"""Ensure TOGGLE_INTENT_MARKERS contains zero localized strings."""
|
||||
|
||||
GERMAN_STRINGS = [
|
||||
"gefällt",
|
||||
"gefolgt",
|
||||
"abonnieren",
|
||||
"speichern",
|
||||
"gespeichert",
|
||||
"antworten",
|
||||
"kommentar",
|
||||
"beitrag",
|
||||
]
|
||||
|
||||
def test_no_german_strings_in_toggle_markers(self):
|
||||
from GramAddict.core.perception.action_memory import TOGGLE_INTENT_MARKERS
|
||||
|
||||
for intent_key, markers in TOGGLE_INTENT_MARKERS.items():
|
||||
for marker in markers:
|
||||
assert marker.lower() not in [
|
||||
g.lower() for g in self.GERMAN_STRINGS
|
||||
], f"TOGGLE_INTENT_MARKERS['{intent_key}'] contains German string '{marker}'!"
|
||||
|
||||
def test_markers_only_contain_english_or_resource_id_patterns(self):
|
||||
"""All markers must be English words or resource_id fragments."""
|
||||
from GramAddict.core.perception.action_memory import TOGGLE_INTENT_MARKERS
|
||||
|
||||
allowed_pattern = re.compile(r"^[a-z_]+$")
|
||||
for intent_key, markers in TOGGLE_INTENT_MARKERS.items():
|
||||
for marker in markers:
|
||||
assert allowed_pattern.match(
|
||||
marker
|
||||
), f"Marker '{marker}' in '{intent_key}' contains non-ASCII or non-ID characters!"
|
||||
|
||||
def test_intent_match_rejects_reel_message_composer(self):
|
||||
"""The semantic guard must reject reel message composer for 'like' intent."""
|
||||
from GramAddict.core.perception.action_memory import _intent_matches_node
|
||||
|
||||
# This was the exact production failure: VLM picked the message composer
|
||||
semantic = "text: 'Send message', desc: '', id: 'com.instagram.android:id/reel_viewer_message_composer_text'"
|
||||
assert _intent_matches_node("tap like button", semantic) is False
|
||||
|
||||
def test_intent_match_accepts_real_like_button(self):
|
||||
"""The semantic guard must accept a real like button by resource_id."""
|
||||
from GramAddict.core.perception.action_memory import _intent_matches_node
|
||||
|
||||
semantic = "text: '', desc: 'Like', id: 'com.instagram.android:id/row_feed_button_like'"
|
||||
assert _intent_matches_node("tap like button", semantic) is True
|
||||
|
||||
def test_intent_match_accepts_like_button_by_id_only(self):
|
||||
"""Even without text/desc, resource_id containing 'like' is enough."""
|
||||
from GramAddict.core.perception.action_memory import _intent_matches_node
|
||||
|
||||
semantic = "text: '', desc: '', id: 'com.instagram.android:id/row_feed_button_like'"
|
||||
assert _intent_matches_node("tap like button", semantic) is True
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════
|
||||
# 2. TelepathicEngine Following Guard must be structural
|
||||
# ══════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestTelepathicEngineFollowingGuardIsStructural:
|
||||
"""The 'already followed' guard must not rely on German strings."""
|
||||
|
||||
def test_no_german_in_following_guard_source(self):
|
||||
"""Scan telepathic_engine.py for any German follow-state strings."""
|
||||
import inspect
|
||||
|
||||
from GramAddict.core.telepathic_engine import TelepathicEngine
|
||||
|
||||
source = inspect.getsource(TelepathicEngine)
|
||||
german_terms = ["gefolgt", "angefragt", "abonniert", "abonnieren"]
|
||||
for term in german_terms:
|
||||
assert (
|
||||
term not in source
|
||||
), f"TelepathicEngine source contains German string '{term}'!"
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════
|
||||
# 3. DarwinEngine comment detection must be structural
|
||||
# ══════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestDarwinEngineCommentDetectionIsStructural:
|
||||
"""The _has_comments heuristic must not rely on German strings."""
|
||||
|
||||
def test_no_german_in_has_comments_source(self):
|
||||
import inspect
|
||||
|
||||
from GramAddict.core.darwin_engine import DarwinEngine
|
||||
|
||||
source = inspect.getsource(DarwinEngine._has_comments)
|
||||
german_terms = ["kommentar", "ansehen"]
|
||||
for term in german_terms:
|
||||
assert (
|
||||
term not in source
|
||||
), f"DarwinEngine._has_comments contains German string '{term}'!"
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════
|
||||
# 4. ResonanceEngine comment filtering must be structural
|
||||
# ══════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestResonanceEngineCommentFilteringIsStructural:
|
||||
"""Comment extraction blocked_exact list must not contain German strings."""
|
||||
|
||||
def test_no_german_in_resonance_source(self):
|
||||
import inspect
|
||||
|
||||
from GramAddict.core.resonance_engine import ResonanceEngine
|
||||
|
||||
source = inspect.getsource(ResonanceEngine.extract_and_learn_comments)
|
||||
german_terms = [
|
||||
"antworten",
|
||||
"gefällt mir",
|
||||
"antworten ansehen",
|
||||
"übersetzung anzeigen",
|
||||
"antworten verbergen",
|
||||
"alle kommentare ansehen",
|
||||
"absenden",
|
||||
"gehe zu",
|
||||
"tippe auf",
|
||||
"aktionen für diesen beitrag",
|
||||
]
|
||||
for term in german_terms:
|
||||
assert (
|
||||
term not in source
|
||||
), f"ResonanceEngine.extract_and_learn_comments contains German '{term}'!"
|
||||
BIN
tests/e2e/__pycache__/__init__.cpython-311.pyc
Normal file
BIN
tests/e2e/__pycache__/__init__.cpython-311.pyc
Normal file
Binary file not shown.
BIN
tests/e2e/__pycache__/conftest.cpython-311-pytest-8.3.5.pyc
Normal file
BIN
tests/e2e/__pycache__/conftest.cpython-311-pytest-8.3.5.pyc
Normal file
Binary file not shown.
BIN
tests/e2e/__pycache__/conftest.cpython-311.pyc
Normal file
BIN
tests/e2e/__pycache__/conftest.cpython-311.pyc
Normal file
Binary file not shown.
BIN
tests/e2e/__pycache__/device_emulator.cpython-311.pyc
Normal file
BIN
tests/e2e/__pycache__/device_emulator.cpython-311.pyc
Normal file
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
tests/e2e/__pycache__/test_debug.cpython-311-pytest-8.3.5.pyc
Normal file
BIN
tests/e2e/__pycache__/test_debug.cpython-311-pytest-8.3.5.pyc
Normal file
Binary file not shown.
Binary file not shown.
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user