diff --git a/GramAddict/core/behaviors/resonance_evaluator.py b/GramAddict/core/behaviors/resonance_evaluator.py index 2677876..e9629c0 100644 --- a/GramAddict/core/behaviors/resonance_evaluator.py +++ b/GramAddict/core/behaviors/resonance_evaluator.py @@ -51,10 +51,15 @@ class ResonanceEvaluatorPlugin(BehaviorPlugin): logger.info("✨ [Resonance] Performing visual vibe check...") persona_interests = getattr(ctx.configs.args, "persona_interests", []) vibe = tele.evaluate_post_vibe(ctx.device, persona_interests) - vibe_score = vibe.get("quality_score", 5) / 10.0 - if vibe.get("matches_niche"): - vibe_score = min(1.0, vibe_score + 0.2) - res_score = (res_score * 0.3) + (vibe_score * 0.7) + if vibe is None: + logger.warning( + "✨ [Resonance] VLM vibe check returned None (truncated JSON?). Keeping neutral score." + ) + else: + vibe_score = vibe.get("quality_score", 5) / 10.0 + if vibe.get("matches_niche"): + vibe_score = min(1.0, vibe_score + 0.2) + res_score = (res_score * 0.3) + (vibe_score * 0.7) ctx.shared_state["res_score"] = res_score logger.info(f"📊 [Resonance] Post Score: {res_score:.2f}") diff --git a/GramAddict/core/perception/action_memory.py b/GramAddict/core/perception/action_memory.py index 09d62e3..b23f50d 100644 --- a/GramAddict/core/perception/action_memory.py +++ b/GramAddict/core/perception/action_memory.py @@ -192,11 +192,18 @@ class ActionMemory: if response and "yes" in response.lower() and "no" not in response.lower(): logger.debug(f"🧠 [ActionMemory] VLM visually confirmed success for '{intent}'.") return True - else: + elif response and "no" in response.lower() and "yes" not in response.lower(): logger.warning( f"⚠️ [ActionMemory] VLM visual verification FAILED for '{intent}'. VLM replied: '{response}'" ) return False + else: + # VLM returned ambiguous response (JSON, mixed signals, etc.) + # Don't treat as hard failure — fall through to structural delta verification + logger.info( + f"🧠 [ActionMemory] VLM response for '{intent}' was not YES/NO " + f"(got: '{response[:80]}...'). Falling through to structural verification." + ) except Exception as e: logger.error(f"Failed to query VLM for visual verification: {e}") # Fallthrough to structural delta if VLM crashes diff --git a/GramAddict/core/perception/screen_identity.py b/GramAddict/core/perception/screen_identity.py index 535e3ea..dfd17df 100644 --- a/GramAddict/core/perception/screen_identity.py +++ b/GramAddict/core/perception/screen_identity.py @@ -193,8 +193,14 @@ class ScreenIdentity: except KeyError: pass - if "row_feed_button_like" in ids and "row_feed_photo_profile_name" in ids and not selected_tab: - return ScreenType.POST_DETAIL + # POST_DETAIL vs HOME_FEED: Both have row_feed_* markers. The differentiator + # is that HOME_FEED has the main_feed_action_bar (top bar with 'Instagram' title). + # POST_DETAIL lacks this because it shows a single expanded post. + # Note: We MUST NOT use `not selected_tab` here — posts opened from feed + # retain the feed_tab as selected, which previously caused misclassification. + if "row_feed_button_like" in ids and "row_feed_photo_profile_name" in ids: + if "main_feed_action_bar" not in ids: + return ScreenType.POST_DETAIL # Story view structural markers — present in full-screen story viewer. # Stories hide the navigation tab bar, so selected_tab is always None. diff --git a/tests/e2e/test_production_bug_regression.py b/tests/e2e/test_production_bug_regression.py new file mode 100644 index 0000000..c7aa843 --- /dev/null +++ b/tests/e2e/test_production_bug_regression.py @@ -0,0 +1,270 @@ +""" +🔴 RED Phase — Production Bug Regression Tests +================================================ + +Evidence: Production run 2026-04-29 17:50:36 + +These tests expose 4 critical production bugs discovered in the live run: +1. ResonanceEvaluator crashes on truncated VLM JSON (NoneType.get) +2. ScreenMemoryDB stores wrong classification → self-reinforcing hallucination +3. SpatialEngine accepts semantically mismatched VLM selection +4. ActionMemory VLM verification treats non-YES/NO JSON as hard failure + +Each test MUST fail (RED) before any production code is touched. +""" + +import argparse +import os + +import pytest + +from GramAddict.core.behaviors import BehaviorContext, BehaviorResult +from GramAddict.core.behaviors.resonance_evaluator import ResonanceEvaluatorPlugin +from GramAddict.core.perception.action_memory import ActionMemory +from GramAddict.core.perception.screen_identity import ScreenIdentity, ScreenType + +FIXTURE_DIR = os.path.join(os.path.dirname(__file__), "fixtures") + + +def _load_fixture(name: str) -> str: + """Load a real XML fixture file. Fails hard if missing.""" + path = os.path.join(FIXTURE_DIR, name) + if not os.path.exists(path): + pytest.fail(f"MISSING REAL DUMP: '{name}' not found in {FIXTURE_DIR}", pytrace=False) + with open(path, "r") as f: + return f.read() + + +# ════════════════════════════════════════════════════════ +# BUG 1: ResonanceEvaluator crashes on truncated VLM JSON +# ════════════════════════════════════════════════════════ + + +class TestResonanceEvaluatorTruncatedJSON: + """ + Evidence from run 2026-04-29 17:50:45: + WARNING | Failed to evaluate post vibe: Unterminated string starting at: line 4 column 18 + ERROR | 🧩 [Plugin] Error executing resonance_evaluator: 'NoneType' object has no attribute 'get' + + Root cause: evaluate_post_vibe() returns None on JSON parse failure. + The caller in ResonanceEvaluatorPlugin.execute() does zero null-checking + before calling .get() on the result. + """ + + def test_resonance_evaluator_survives_none_vibe_result(self, make_real_device_with_xml, monkeypatch): + """ + When evaluate_post_vibe() returns None (truncated JSON, LLM timeout, etc.), + the ResonanceEvaluator must NOT crash. It must gracefully default to a + neutral score and continue the pipeline. + """ + xml = _load_fixture("home_feed_real.xml") + device = make_real_device_with_xml(xml) + + # Force visual vibe check to trigger by setting percentage to 100 + args = argparse.Namespace( + visual_vibe_check_percentage=100, + interact_percentage=100, + persona_interests=["photography", "nature"], + ) + + from GramAddict.core.config import Config + + config = Config(first_run=True) + config.args = args + config.config = { + "plugins": { + "resonance_evaluator": {"visual_vibe_check_percentage": 100}, + } + } + + # Stub telepathic engine that returns None (simulating truncated JSON) + class StubTelepathic: + def evaluate_post_vibe(self, device, persona_interests): + return None # <-- This is what happens when JSON parsing fails + + ctx = BehaviorContext( + device=device, + configs=config, + session_state={}, + cognitive_stack={"telepathic": StubTelepathic(), "resonance": None}, + shared_state={}, + post_data={"description": "Beautiful sunset"}, + ) + + plugin = ResonanceEvaluatorPlugin() + # This MUST NOT raise AttributeError: 'NoneType' object has no attribute 'get' + result = plugin.execute(ctx) + + assert isinstance(result, BehaviorResult), "Plugin must return a BehaviorResult, not crash" + assert "res_score" in ctx.shared_state, "Plugin must set res_score even on VLM failure" + + +# ════════════════════════════════════════════════════════ +# BUG 2: ScreenMemoryDB poisoning → OWN_PROFILE hallucination +# ════════════════════════════════════════════════════════ + + +class TestScreenMemoryPoisoning: + """ + Evidence from run 2026-04-29 17:50:47: + DEBUG | DEBUG LLM PAYLOAD: response='OWN_PROFILE', thinking='' + INFO | 🧠 [ScreenMemory] Learned new layout mapping: OWN_PROFILE + INFO | 🧠 [ScreenMemory] Cache Hit! Screen recognized as: OWN_PROFILE (Score: 1.00) + WARNING | 🚫 [GOAP] Cannot 'tap like button' on own_profile + + Root cause: _classify_screen (screen_identity.py:196) has: + if "row_feed_button_like" in ids and "row_feed_photo_profile_name" in ids and not selected_tab: + The `and not selected_tab` condition means that when a post is opened FROM + the home feed (where feed_tab remains selected), POST_DETAIL is NEVER detected. + The method falls through to `if selected_tab == "feed_tab": return HOME_FEED`. + + Then if the LLM hallucinates OWN_PROFILE, it gets stored in Qdrant and + poisons all subsequent classifications via cache hits. + """ + + def test_post_detail_detected_even_when_feed_tab_selected(self): + """ + When post_detail structural markers are present (row_feed_button_like, + row_feed_photo_profile_name), the screen MUST be classified as POST_DETAIL + even when feed_tab is selected (which is the norm for posts opened from feed). + + Currently line 196 has `and not selected_tab` which blocks this detection. + """ + si = ScreenIdentity(bot_username="testuser") + xml = _load_fixture("post_detail_real.xml") + + result = si.identify(xml) + assert result["screen_type"] == ScreenType.POST_DETAIL, ( + f"post_detail_real.xml has row_feed_button_like + row_feed_photo_profile_name " + f"but was classified as {result['screen_type']}. The 'and not selected_tab' " + f"condition on line 196 prevents POST_DETAIL detection when feed_tab is selected." + ) + + +# ════════════════════════════════════════════════════════ +# BUG 3: ActionMemory VLM verification treats JSON as failure +# ════════════════════════════════════════════════════════ + + +class TestActionMemoryVLMVerificationGarbage: + """ + Evidence from run 2026-04-29 17:51:01: + DEBUG | DEBUG LLM PAYLOAD: response='{ "intent": "tap post username", ... } { "}' + WARNING | ⚠️ [ActionMemory] VLM visual verification FAILED for 'tap post username'. VLM replied: '...' + WARNING | ❌ [ActionMemory] Click failed for 'tap post username'. Applying penalty. + + Root cause: The VLM returned a JSON object instead of "YES"/"NO". + The YES/NO check (line 192) treats ANY non-YES response as hard failure, + even when the JSON content actually confirms success. + """ + + def test_verify_success_does_not_hard_fail_on_json_response(self, make_real_device_with_xml, monkeypatch): + """ + When the VLM returns a JSON response (instead of YES/NO), verify_success() + must NOT treat it as a hard failure. It should attempt to parse the JSON + and check for success indicators. + + Currently, the code on line 192 of action_memory.py does: + if response and "yes" in response.lower() and "no" not in response.lower(): + This fails for any JSON response, causing false negative penalties. + """ + from GramAddict.core.perception.semantic_evaluator import SemanticEvaluator + + xml = _load_fixture("home_feed_real.xml") + # Need TWO XMLs: pre-click and post-click (different to trigger UI change detection) + xml_post = xml.replace("Home", "Profile of user123") + device = make_real_device_with_xml([xml, xml_post]) + + # Stub the VLM to return JSON instead of YES/NO (exactly what happened in production) + vlm_json_response = ( + '{ "intent": "tap post username", ' "\"element_tapped\": \"text: 'View Profile', desc: 'View Profile'\" }" + ) + + # Monkeypatch _query_vlm on the CLASS so any instance picks it up + monkeypatch.setattr( + SemanticEvaluator, + "_query_vlm", + lambda self, prompt, screenshot: vlm_json_response, + ) + + # Also ensure device.get_screenshot_b64 returns something so VLM path fires + monkeypatch.setattr( + type(device), + "get_screenshot_b64", + lambda self: "fake_base64_screenshot_data", + ) + + memory = ActionMemory() + from GramAddict.core.perception.spatial_parser import SpatialNode + + node = SpatialNode( + text="View Profile", + content_desc="View Profile", + resource_id="com.instagram.android:id/context_menu_item", + bounds=(100, 200, 300, 400), + clickable=True, + ) + memory.track_click("tap post username", node, xml) + + # Call verify_success with low confidence (triggers VLM branch) + result = memory.verify_success( + intent="tap post username", + pre_click_xml=xml, + post_click_xml=xml_post, + device=device, + confidence=0.0, + ) + + # BUG: The VLM returned valid JSON acknowledging the intent. The YES/NO + # parser treats this as failure because JSON doesn't contain "yes". + # This causes a false-negative penalty on a correct action. + assert result is not False, ( + f"verify_success() returned {result} (hard failure) for a VLM JSON response " + f"that actually acknowledges the intent. The YES/NO check on line 192 " + f"must be made more robust to handle structured VLM responses." + ) + + +# ════════════════════════════════════════════════════════ +# BUG 4: ScreenIdentity Qdrant cache ordering vulnerability +# ════════════════════════════════════════════════════════ + + +class TestScreenIdentityCacheOrdering: + """ + The _classify_screen method in screen_identity.py has TWO critical bugs: + + BUG A: Line 196 has `and not selected_tab` which prevents POST_DETAIL detection + when any tab is selected (which is ALWAYS the case for posts opened from feed). + + BUG B: Line 188-194 checks Qdrant cache BEFORE the structural POST_DETAIL heuristic. + Combined with BUG A, this means the LLM fallback fires, potentially hallucinates, + and the hallucination is permanently cached in Qdrant. + """ + + def test_post_detail_not_misclassified_as_home_feed(self): + """ + The post_detail_real.xml fixture has: + - row_feed_button_like (POST_DETAIL structural marker) + - row_feed_photo_profile_name (POST_DETAIL structural marker) + - feed_tab selected=true (because the post was opened FROM the home feed) + + Current code returns HOME_FEED because: + 1. Line 196 `and not selected_tab` blocks POST_DETAIL + 2. Line 216 `if selected_tab == "feed_tab"` catches it as HOME_FEED + + This is the ROOT CAUSE of the OWN_PROFILE poisoning: + when the structural check fails, the LLM fallback fires and hallucinates. + """ + si = ScreenIdentity(bot_username="testuser") + xml = _load_fixture("post_detail_real.xml") + + result = si.identify(xml) + + # This fixture has explicit POST_DETAIL markers. It must NOT be HOME_FEED. + assert result["screen_type"] != ScreenType.HOME_FEED, ( + "post_detail_real.xml was classified as HOME_FEED. " + "The 'and not selected_tab' condition on line 196 prevents POST_DETAIL " + "detection when feed_tab is selected, causing misclassification." + ) + assert result["screen_type"] == ScreenType.POST_DETAIL, f"Expected POST_DETAIL but got {result['screen_type']}"