From c690ca589819d5e4ce69015d084161f65ea8e3e8 Mon Sep 17 00:00:00 2001 From: jmoreira-valory Date: Sat, 5 Sep 2026 18:23:52 +0200 Subject: [PATCH 1/2] fix: port the issue-455 free-text contract fix to 9 sibling tools (lockstep) Every tool that still derived its web-search query from the trader-template regex with a raw-prompt fallback gets the merged #459+#461 machinery: byte-identical parse_prompt block (frozen mirror, now 11 files), _flagged_null_result on empty retrieval, unsearchable-query short-circuit, and scan_truncated/tier observability. Trader-template path byte-identical; free text keeps the whole prompt for the LLM and derives only the query. In scope: superforcaster, superforcaster_calibrated_full_search, superforcaster_full_search, superforcaster_polymarket_v1/v2/v3, finetuned_prediction (tournament pins re-pinned), prediction_request_rag_v1, prediction_request_reasoning_v1. Out of scope (different fix shape): factual_research family and the other partial-leak tools. 945 customs tests green; sextet green; lock --check verified. Co-Authored-By: Claude Fable 5 --- benchmark/tournament_tools.json | 6 +- .../prediction_request_rag_v1/component.yaml | 4 +- .../prediction_request_rag_v1.py | 403 ++++++++++++++++-- .../tests/test_prediction_request_rag_v1.py | 272 +++++++++++- .../component.yaml | 4 +- .../prediction_request_reasoning_v1.py | 368 +++++++++++++++- .../test_prediction_request_reasoning_v1.py | 355 ++++++++++++++- packages/packages.json | 22 +- .../agents/mech_predict/aea-config.yaml | 16 +- .../finetuned_prediction/component.yaml | 4 +- .../finetuned_prediction.py | 353 +++++++++++++-- .../tests/test_finetuned_prediction.py | 149 ++++++- .../customs/superforcaster/component.yaml | 4 +- .../customs/superforcaster/superforcaster.py | 350 ++++++++++++++- .../tests/test_superforcaster.py | 196 +++++++++ .../component.yaml | 4 +- .../superforcaster_calibrated_full_search.py | 361 +++++++++++++++- ...t_superforcaster_calibrated_full_search.py | 251 +++++++++++ .../superforcaster_full_search/component.yaml | 4 +- .../superforcaster_full_search.py | 357 +++++++++++++++- .../tests/test_superforcaster_full_search.py | 209 +++++++++ .../component.yaml | 4 +- .../superforcaster_polymarket_v1.py | 346 ++++++++++++++- .../tests/test_superforcaster.py | 197 +++++++++ .../component.yaml | 4 +- .../superforcaster_polymarket_v2.py | 346 ++++++++++++++- .../tests/test_superforcaster.py | 217 ++++++++++ .../component.yaml | 5 +- .../superforcaster_polymarket_v3.py | 347 ++++++++++++++- .../tests/test_superforcaster.py | 16 +- .../test_superforcaster_polymarket_v3.py | 374 ++++++++++++++++ .../valory/services/mech_predict/service.yaml | 2 +- 32 files changed, 5292 insertions(+), 258 deletions(-) create mode 100644 packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster_polymarket_v3.py diff --git a/benchmark/tournament_tools.json b/benchmark/tournament_tools.json index f31e61a67..d6d7245d1 100644 --- a/benchmark/tournament_tools.json +++ b/benchmark/tournament_tools.json @@ -1,8 +1,8 @@ { "factual_research-v2": "bafybeidgo3jesmvuwa64ooij7mywk7hf7ls5len2sf7cbcjwt3t6r5u6qm", - "predict-base": "bafybeiclpkn4sqvri7k5aiklt5fm6hnt3tgo4qxtrkzxe4fwzfs2plgbk4", - "predict-fine-tuned": "bafybeiclpkn4sqvri7k5aiklt5fm6hnt3tgo4qxtrkzxe4fwzfs2plgbk4", - "predict-fine-tuned-calibrated": "bafybeiclpkn4sqvri7k5aiklt5fm6hnt3tgo4qxtrkzxe4fwzfs2plgbk4", + "predict-base": "bafybeifyhwavsm72okmmibc4yhxnlys5gcouhiydvathagpwnl5sgevsra", + "predict-fine-tuned": "bafybeifyhwavsm72okmmibc4yhxnlys5gcouhiydvathagpwnl5sgevsra", + "predict-fine-tuned-calibrated": "bafybeifyhwavsm72okmmibc4yhxnlys5gcouhiydvathagpwnl5sgevsra", "superforcaster-polymarket-v4": "bafybeiefu5cetnebkr2yza6yn6la2ldkwvn2pjlgcmew2e6fblnyjh5roq", "superforcaster-market-aware": "bafybeiecvkdx2zjezc5aslnpg4nbw25rsbrc7xtpywbkjv3d47tdvskhgq" } diff --git a/packages/napthaai/customs/prediction_request_rag_v1/component.yaml b/packages/napthaai/customs/prediction_request_rag_v1/component.yaml index 6a4b90f5b..12685aae2 100644 --- a/packages/napthaai/customs/prediction_request_rag_v1/component.yaml +++ b/packages/napthaai/customs/prediction_request_rag_v1/component.yaml @@ -7,9 +7,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeifnw5qoyshsiq2g7ulz4q3vbjpy7pxgxpop3f5dsrnfgm5w3dahz4 - prediction_request_rag_v1.py: bafybeifucm27pdzmdfxpyjszabclmpkirk5jbjq5zwdmnw5rdvtv7ehjlq + prediction_request_rag_v1.py: bafybeibr34vvrq2s4morz4tr4ahn7vfuc2gegtow6rkgs44jdphfd37sx4 tests/__init__.py: bafybeifcgilmgfwx7kaap67cfnuokrhfrabi6bnvqiudyzgf2idu64dvxq - tests/test_prediction_request_rag_v1.py: bafybeibe7id3s5arysqxkmongbqhep6olcfvxpabklmuuoealskyvpnnay + tests/test_prediction_request_rag_v1.py: bafybeiayzev7seunarcbociixwqqqrajraoc3dflgesmlmgdr4o4tjp5rq fingerprint_ignore_patterns: [] entry_point: prediction_request_rag_v1.py callable: run diff --git a/packages/napthaai/customs/prediction_request_rag_v1/prediction_request_rag_v1.py b/packages/napthaai/customs/prediction_request_rag_v1/prediction_request_rag_v1.py index 806587e44..342ca8c1f 100644 --- a/packages/napthaai/customs/prediction_request_rag_v1/prediction_request_rag_v1.py +++ b/packages/napthaai/customs/prediction_request_rag_v1/prediction_request_rag_v1.py @@ -24,7 +24,18 @@ import re from concurrent.futures import Future, ThreadPoolExecutor from io import BytesIO -from typing import Any, Callable, Dict, Generator, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + Generator, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import anthropic import faiss @@ -67,6 +78,9 @@ GOOGLE_RATE_LIMIT_EXCEEDED_CODE = 429 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 def with_key_rotation(func: Callable) -> Callable: @@ -306,6 +320,8 @@ def embeddings(self, model: Any, input_: Any) -> Any: DEFAULT_NUM_URLS = 3 DEFAULT_NUM_QUERIES = 2 NUM_URLS_PER_QUERY = 5 +# Per-query cap on Serper organic results (see _shape_serper_sources). +MAX_SOURCES = 5 SPLITTER_CHUNK_SIZE = 1800 SPLITTER_OVERLAP = 50 EMBEDDING_MODEL = "text-embedding-3-large" @@ -392,6 +408,15 @@ def embeddings(self, model: Any, input_: Any) -> Any: SYSTEM_PROMPT = """You are a world class algorithm for generating structured output from a given input.""" +class EmptyRetrievalError(ValueError): + """Raised when retrieval yields no usable documents (issue #455). + + Distinct from the typed shape errors raised by _shape_serper_sources: a + genuine zero-hit converges on the flagged null in run(), while a broken + or reshaped integration surfaces as an error null with error_type. + """ + + class ExtendedDocument(BaseModel): """Document model""" @@ -482,10 +507,13 @@ def multi_queries( counter_callback: Optional[Callable] = None, temperature: float = LLM_SETTINGS["claude-sonnet-4-6"]["temperature"], max_tokens: int = LLM_SETTINGS["claude-sonnet-4-6"]["default_max_tokens"], + search_query: Optional[str] = None, ) -> Tuple[List[str], Optional[Callable]]: """Generate multiple queries for fetching information from the web.""" if not client: raise RuntimeError("Client not initialized") + if search_query is None: + search_query = prompt url_query_prompt = URL_QUERY_PROMPT.format( USER_PROMPT=prompt, NUM_QUERIES=num_queries @@ -516,7 +544,10 @@ def multi_queries( queries = [query for query in queries if query.strip() != ""] if len(queries) > DEFAULT_NUM_QUERIES: queries = queries[:DEFAULT_NUM_QUERIES] - queries.append(prompt) + # The direct query sent alongside the brainstormed ones is the compressed + # search query, not the raw prompt: a prompt-shaped query degrades Serper + # sharply, in the worst case to zero organic results (issue #455). + queries.append(search_query) return queries, counter_callback @@ -604,8 +635,13 @@ def get_urls_from_queries_serper( ) response.raise_for_status() data = response.json() - organic = data.get("organic", []) + organic, _ = _shape_serper_sources(data, "live search") urls.extend(item["link"] for item in organic[:num]) + except ValueError: + # A missing/malformed organic key is a broken or reshaped + # integration (a quota-error body hits every query alike), not a + # zero-hit -- surface it as an error null instead of swallowing. + raise except Exception as e: print(f"Error fetching URLs for query '{query}': {e}") return list(set(urls)) @@ -845,7 +881,7 @@ def recursive_character_text_splitter( return [text[i : i + max_tokens] for i in range(0, len(text), max_tokens - overlap)] -def fetch_additional_information( # pylint: disable=too-many-statements +def fetch_additional_information( # pylint: disable=too-many-statements,too-many-locals client: "LLMClient", client_embedding: Optional["LLMClient"], prompt: str, @@ -861,9 +897,12 @@ def fetch_additional_information( # pylint: disable=too-many-statements num_queries: int = DEFAULT_NUM_QUERIES, temperature: float = LLM_SETTINGS["claude-sonnet-4-6"]["temperature"], max_tokens: int = LLM_SETTINGS["claude-sonnet-4-6"]["default_max_tokens"], + search_query: Optional[str] = None, ) -> Tuple[str, Dict[str, Any], Optional[Callable[..., None]]]: """Fetch additional information to help answer the user prompt.""" # generate multiple queries for fetching information from the web + if search_query is None: + search_query = prompt try: queries, counter_callback = multi_queries( @@ -874,11 +913,12 @@ def fetch_additional_information( # pylint: disable=too-many-statements counter_callback=counter_callback, temperature=temperature, max_tokens=max_tokens, + search_query=search_query, ) print(f"Queries: {queries}") except Exception as e: print(f"Error generating queries: {e}") - queries = [prompt] + queries = [search_query] # get the top URLs for the queries if source_content is None: @@ -952,7 +992,10 @@ def fetch_additional_information( # pylint: disable=too-many-statements print(f"Split Docs: {len(split_docs)}") if len(split_docs) == 0: - raise ValueError("No valid documents found from the provided URLs") + # Retrieval came back empty (zero organic hits, or every page failed + # to fetch/extract). run() converges this on the flagged null rather + # than an opaque error string (issue #455). + raise EmptyRetrievalError("No valid documents found from the provided URLs") if len(split_docs) > MAX_NR_DOCS: # truncate the split_docs to the first MAX_NR_DOCS documents @@ -981,17 +1024,248 @@ def fetch_additional_information( # pylint: disable=too-many-statements return additional_information, raw_source_content, counter_callback -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: - print(f"Error extracting question: {e}") - question = prompt +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc - return question + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). The on-chain JSON carries + only the four standard fields; the explicit marker for requesters lives in + used_params["empty_retrieval"] (off-chain metadata.params). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped search ("empty query") stays + distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[prediction-request-rag-v1] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params def parser_prediction_response(response: str) -> str: @@ -1013,7 +1287,7 @@ def parser_prediction_response(response: str) -> str: @with_key_rotation -def run( +def run( # pylint: disable=too-many-locals **kwargs: Any, ) -> Union[float, MechResponse]: """Run the task""" @@ -1045,7 +1319,30 @@ def run( llm_client, embedding_client, ): - prompt = extract_question(kwargs["prompt"]) + prompt = kwargs["prompt"] + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[prediction-request-rag-v1] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[prediction-request-rag-v1] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[prediction-request-rag-v1] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) max_tokens = kwargs.get("max_tokens", LLM_SETTINGS[model]["default_max_tokens"]) temperature = kwargs.get("temperature", LLM_SETTINGS[model]["temperature"]) num_urls = kwargs.get("num_urls", DEFAULT_NUM_URLS) @@ -1073,30 +1370,68 @@ def run( raise ValueError( f"Invalid source_content_mode: {source_content_mode!r}. Must be 'cleaned' or 'raw'." ) - additional_information, source_content, counter_callback = ( - fetch_additional_information( - client=llm_client, - client_embedding=embedding_client, - prompt=prompt, + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # query-brainstorm and search calls and return the flagged null + # directly. + return _flagged_null_result( model=model, - google_api_key=google_api_key, - google_engine_id=google_engine_id, - serper_api_key=serper_api_key, - search_provider=search_provider, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, counter_callback=counter_callback, - source_content=kwargs.get("source_content", None), - source_content_mode=source_content_mode, - num_urls=num_urls, - num_queries=num_queries, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) + cached_source_content = kwargs.get("source_content", None) + try: + additional_information, source_content, counter_callback = ( + fetch_additional_information( + client=llm_client, + client_embedding=embedding_client, + prompt=question, + model=model, + google_api_key=google_api_key, + google_engine_id=google_engine_id, + serper_api_key=serper_api_key, + search_provider=search_provider, + counter_callback=counter_callback, + source_content=cached_source_content, + source_content_mode=source_content_mode, + num_urls=num_urls, + num_queries=num_queries, + temperature=temperature, + max_tokens=max_tokens, + search_query=search_query, + ) + ) + except EmptyRetrievalError: + # Retrieval came back empty (zero hits live, or an empty cached + # capture on replay) -- return the flagged null instead of an + # opaque error string (issue #455). + return _flagged_null_result( + model=model, temperature=temperature, max_tokens=max_tokens, + captured_source_content=cached_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context=( + "cached replay" + if cached_source_content is not None + else "live search" + ), + tier=tier, + scan_truncated=scan_truncated, ) - ) # Generate the prediction prompt prediction_prompt = PREDICTION_PROMPT.format( ADDITIONAL_INFORMATION=additional_information, - USER_PROMPT=prompt, + USER_PROMPT=question, ) # Generate the prediction @@ -1117,6 +1452,8 @@ def run( "max_tokens": max_tokens, "num_urls": num_urls, "num_queries": num_queries, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = source_content diff --git a/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py b/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py index cd1411497..4083a35cd 100644 --- a/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py +++ b/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py @@ -20,6 +20,7 @@ """Unit tests for prediction_request_rag: thread-safe client, offline tiktoken, and source_content.""" import inspect +import json from concurrent.futures import Future from pathlib import Path from types import SimpleNamespace @@ -376,7 +377,9 @@ def test_empty_source_content_raises(self, mock_queries: MagicMock) -> None: ) -def _make_mock_api_keys(return_source_content: str = "false") -> MagicMock: +def _make_mock_api_keys( + return_source_content: str = "false", **overrides: Any +) -> MagicMock: """Create a mock api_keys object (KeyChain-like) for run().""" services = { "openai": "sk-test", @@ -386,6 +389,7 @@ def _make_mock_api_keys(return_source_content: str = "false") -> MagicMock: "search_provider": "google", "return_source_content": return_source_content, } + services.update(overrides) mock_keys = MagicMock() mock_keys.__getitem__ = MagicMock(side_effect=lambda k: services[k]) mock_keys.get = MagicMock( @@ -645,3 +649,269 @@ def test_anthropic_tokenizer_error_falls_back(self) -> None: result = count_tokens("hello world", "claude-sonnet-4-6", client=mock_client) assert isinstance(result, int) assert result > 0 + + +# --------------------------------------------------------------------------- +# issue-455 free-text input contract, ported from superforcaster-polymarket-v4: +# parse_prompt tiers, the no-alphanumeric short-circuit, the empty-retrieval +# flagged null, and the typed Serper shape errors. +# --------------------------------------------------------------------------- + +# Trader-template format prompt (regression: previous callers must still work) +TRADER_PROMPT = ( + 'Given the question "Will X happen?" and the `yes` answer criterion, ...' +) +# Free-text prompt that would return degraded Serper results if passed raw +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently transfer " + "to Liverpool FC before the end of the summer 2025 transfer window (September 2, " + "2025 23:59 UTC)? Resolution source: official club announcements or BBC Sport. " + "The market resolves YES if a permanent transfer (not a loan) is confirmed by " + "the resolution source before the deadline." +) + + +def _mock_client_manager(mock_mgr: MagicMock) -> tuple: + """Configure a mocked LLMClientManager and return its (llm, embed) pair.""" + mock_llm = MagicMock() + mock_embed = MagicMock() + mock_mgr.return_value.__enter__ = MagicMock(return_value=(mock_llm, mock_embed)) + mock_mgr.return_value.__exit__ = MagicMock(return_value=False) + return mock_llm, mock_embed + + +VALID_TAGGED_COMPLETION = ( + "0.50.50.5" + "0.5" +) + + +class TestParsePromptContract: + """parse_prompt() -> (question_for_llm, search_query, tier).""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = module.parse_prompt(TRADER_PROMPT) + assert question == "Will X happen?" + assert query == question + assert tier == "template" + + def test_free_text_llm_gets_full_prompt(self) -> None: + """Free-text input: the LLM question is the whole prompt.""" + question, _, tier = module.parse_prompt(LONG_FREE_TEXT_PROMPT) + assert question == LONG_FREE_TEXT_PROMPT + assert tier == "clause" + + def test_boilerplate_prefix_is_dropped_from_query(self) -> None: + """The query anchors at the market question, dropping the lead-in.""" + _, query, _ = module.parse_prompt(LONG_FREE_TEXT_PROMPT) + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert ( + len(query) <= module._MAX_SEARCH_QUERY_LEN + ) # pylint: disable=protected-access + + +class TestDegenerateShortCircuit: + """Prompts with nothing searchable never reach the brainstorm or search.""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{RAG_MODULE}.get_urls_from_queries_serper") + @patch(f"{RAG_MODULE}.get_urls_from_queries") + @patch(f"{RAG_MODULE}.multi_queries") + @patch(f"{RAG_MODULE}.LLMClientManager") + def test_degenerate_prompt_short_circuits_with_zero_search_calls( + self, + mock_mgr: MagicMock, + mock_queries: MagicMock, + mock_google: MagicMock, + mock_serper: MagicMock, + degenerate: str, + ) -> None: + """Degenerate prompts return the flagged null before any network call.""" + _mock_client_manager(mock_mgr) + result = run( + tool="prediction-request-rag-v1", + model="gpt-4.1-2025-04-14", + prompt=degenerate, + api_keys=_make_mock_api_keys(), + ) + mock_queries.assert_not_called() + mock_google.assert_not_called() + mock_serper.assert_not_called() + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "empty query" + assert result[4]["scan_truncated"] is False + + +class TestEmptyRetrievalFlaggedNull: + """Empty retrieval converges on the flagged null, not an error string.""" + + @patch(f"{RAG_MODULE}.multi_queries", return_value=(["market question"], None)) + @patch(f"{RAG_MODULE}.LLMClientManager") + def test_zero_hit_null_reason_is_live_search( + self, mock_mgr: MagicMock, mock_queries: MagicMock + ) -> None: + """A genuine zero-hit records null_reason='live search'.""" + _mock_client_manager(mock_mgr) + serper_resp = MagicMock() + serper_resp.raise_for_status.return_value = None + serper_resp.json.return_value = {"organic": [], "peopleAlsoAsk": []} + with patch(f"{RAG_MODULE}.requests.request", return_value=serper_resp): + result = run( + tool="prediction-request-rag-v1", + model="gpt-4.1-2025-04-14", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys( + search_provider="serper", serperapi="serper-test" + ), + ) + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "live search" + assert result[4]["parse_tier"] == "clause" + + @patch(f"{RAG_MODULE}.multi_queries", return_value=(["market question"], None)) + @patch(f"{RAG_MODULE}.LLMClientManager") + def test_empty_cached_replay_null_reason_is_cached_replay( + self, mock_mgr: MagicMock, mock_queries: MagicMock + ) -> None: + """An empty cached capture on replay records null_reason='cached replay'.""" + _mock_client_manager(mock_mgr) + result = run( + tool="prediction-request-rag-v1", + model="gpt-4.1-2025-04-14", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + source_content={"pages": {}, "pdfs": {}}, + ) + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "cached replay" + + def test_malformed_serper_body_raises_typed_error(self) -> None: + """A missing/malformed organic key raises instead of being swallowed.""" + serper_resp = MagicMock() + serper_resp.raise_for_status.return_value = None + serper_resp.json.return_value = {"organic": None} + with patch(f"{RAG_MODULE}.requests.request", return_value=serper_resp): + with pytest.raises(ValueError, match="organic"): + module.get_urls_from_queries_serper(["q"], api_key="k", num=5) + + def test_empty_serper_body_returns_no_urls(self) -> None: + """A well-formed zero-hit body yields no URLs without raising.""" + serper_resp = MagicMock() + serper_resp.raise_for_status.return_value = None + serper_resp.json.return_value = {"organic": [], "peopleAlsoAsk": []} + with patch(f"{RAG_MODULE}.requests.request", return_value=serper_resp): + assert ( + module.get_urls_from_queries_serper(["q"], api_key="k", num=5) == [] + ) # pylint: disable=use-implicit-booleaness-not-comparison + + +class TestRunParityAndParseMetadata: + """run() wiring: LLM-input parity on the template path + parse metadata.""" + + @staticmethod + def _run_with_fetch_mock(prompt: str) -> tuple: + """Run the tool with fetch + LLM mocked; return (result, fetch kwargs).""" + with ( + patch(f"{RAG_MODULE}.LLMClientManager") as mock_mgr, + patch(f"{RAG_MODULE}.fetch_additional_information") as mock_fetch, + ): + mock_llm, _ = _mock_client_manager(mock_mgr) + mock_fetch.return_value = ("additional info", {"pages": {}}, None) + mock_llm.completions.return_value = MagicMock( + content=VALID_TAGGED_COMPLETION, + usage=MagicMock(prompt_tokens=10, completion_tokens=5), + ) + result = run( + tool="prediction-request-rag-v1", + model="gpt-4.1-2025-04-14", + prompt=prompt, + api_keys=_make_mock_api_keys(), + ) + return result, mock_fetch.call_args.kwargs + + def test_trader_template_feeds_extracted_question_everywhere(self) -> None: + """LLM-input parity: template path is byte-identical to extract_question.""" + result, fetch_kwargs = self._run_with_fetch_mock(TRADER_PROMPT) + assert fetch_kwargs["prompt"] == "Will X happen?" + assert fetch_kwargs["search_query"] == "Will X happen?" + assert "Will X happen?" in result[1] + assert result[4]["parse_tier"] == "template" + + def test_long_template_prompt_is_not_marked_truncated(self) -> None: + """Template past the scan window is NOT flagged as truncated.""" + prompt = TRADER_PROMPT + " filler" * ( + module._MAX_SCAN_CHARS // 3 + ) # pylint: disable=protected-access + assert len(prompt) > module._MAX_SCAN_CHARS # pylint: disable=protected-access + result, _ = self._run_with_fetch_mock(prompt) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + def test_free_text_llm_receives_full_prompt_and_short_query(self) -> None: + """Free text: the LLM sees the whole prompt; search gets the clause.""" + result, fetch_kwargs = self._run_with_fetch_mock(LONG_FREE_TEXT_PROMPT) + assert "official club announcements or BBC Sport" in result[1] + assert fetch_kwargs["prompt"] == LONG_FREE_TEXT_PROMPT + assert fetch_kwargs["search_query"].startswith("Will Alexander Isak") + + +class TestSearchQueryPlumbing: + """The compressed query replaces the raw prompt at the direct-search site.""" + + def test_multi_queries_appends_search_query_not_prompt(self) -> None: + """The direct query appended to the brainstormed ones is search_query.""" + client = MagicMock() + client.completions.return_value = MagicMock( + content="\nquery one\nquery two\n", + usage=MagicMock(prompt_tokens=1, completion_tokens=1), + ) + queries, _ = multi_queries( + client=client, + prompt="LONG PROMPT", + model="gpt-4.1-2025-04-14", + num_queries=2, + search_query="short q", + ) + assert queries[-1] == "short q" + assert "LONG PROMPT" not in queries + + def test_multi_queries_defaults_to_prompt_without_search_query(self) -> None: + """Without a search_query, the old append-the-prompt behavior holds.""" + client = MagicMock() + client.completions.return_value = MagicMock( + content="\nquery one\nquery two\n", + usage=MagicMock(prompt_tokens=1, completion_tokens=1), + ) + queries, _ = multi_queries( + client=client, + prompt="LONG PROMPT", + model="gpt-4.1-2025-04-14", + num_queries=2, + ) + assert queries[-1] == "LONG PROMPT" + + @patch(f"{RAG_MODULE}.get_urls_from_queries_serper", return_value=[]) + @patch(f"{RAG_MODULE}.multi_queries", side_effect=RuntimeError("boom")) + def test_brainstorm_failure_falls_back_to_search_query( + self, mock_queries: MagicMock, mock_serper: MagicMock + ) -> None: + """When the brainstorm fails, the fallback query is search_query.""" + with pytest.raises(module.EmptyRetrievalError): + fetch_additional_information( + client=MagicMock(), + client_embedding=MagicMock(), + prompt=LONG_FREE_TEXT_PROMPT, + model="gpt-4.1-2025-04-14", + google_api_key=None, + google_engine_id=None, + serper_api_key="k", + search_provider="serper", + search_query="short q", + ) + mock_serper.assert_called_once() + assert mock_serper.call_args.kwargs["queries"] == ["short q"] diff --git a/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml b/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml index 4d4672b77..e8187cd3c 100644 --- a/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml +++ b/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml @@ -8,9 +8,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeibcbvmr7v5n2vintaunxjix2jxopbvjij5hytewh3xtlca4mfwhky - prediction_request_reasoning_v1.py: bafybeifk7lx2iy2n54dh4zairovbfjhbz4i5bmdwnfsnjm7tuu7uanq43i + prediction_request_reasoning_v1.py: bafybeifjrnrdx47s2n3bwggblf5bwa7cgkr3qq2emy2tgyx7twera6gmfu tests/__init__.py: bafybeieu3toasuyaehkqounrtylk5bjer7qgvnt6ccjcsi6jlfig7wfrka - tests/test_prediction_request_reasoning_v1.py: bafybeifqnlbc5iutq2sbvhvwoilrbzypmg2bt32raytxcnff46gaqcjcqu + tests/test_prediction_request_reasoning_v1.py: bafybeid4acfxihfnm6l7usi4sy7cjkgdnfclxny6u5ysyjsf2fqyfdiafq fingerprint_ignore_patterns: [] entry_point: prediction_request_reasoning_v1.py callable: run diff --git a/packages/napthaai/customs/prediction_request_reasoning_v1/prediction_request_reasoning_v1.py b/packages/napthaai/customs/prediction_request_reasoning_v1/prediction_request_reasoning_v1.py index 0080fa7b3..723b1a622 100644 --- a/packages/napthaai/customs/prediction_request_reasoning_v1/prediction_request_reasoning_v1.py +++ b/packages/napthaai/customs/prediction_request_reasoning_v1/prediction_request_reasoning_v1.py @@ -26,7 +26,18 @@ import time from concurrent.futures import Future, ThreadPoolExecutor from io import BytesIO -from typing import Any, Callable, Dict, Generator, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + Generator, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import anthropic import faiss @@ -70,6 +81,9 @@ USER_AGENT_HEADER = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36" GOOGLE_RATE_LIMIT_EXCEEDED_CODE = 429 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 def get_model_encoding(model: str) -> Encoding: @@ -308,6 +322,7 @@ def embeddings(self, model: Any, input_: Any) -> Any: ALLOWED_MODELS = list(LLM_SETTINGS.keys()) DEFAULT_NUM_URLS = 3 DEFAULT_NUM_QUERIES = 2 +MAX_SOURCES = 5 SPLITTER_CHUNK_SIZE = 300 SPLITTER_OVERLAP = 50 EMBEDDING_MODEL = "text-embedding-3-large" @@ -529,6 +544,7 @@ def parser_prediction_response(response: str) -> str: def multi_queries( client: "LLMClient", prompt: str, + search_query: str, model: str, num_queries: int, counter_callback: Optional[Callable] = None, @@ -566,7 +582,10 @@ def multi_queries( queries = [query for query in queries if query.strip() != ""] if len(queries) > DEFAULT_NUM_QUERIES: queries = queries[:DEFAULT_NUM_QUERIES] - queries.append(prompt) + # Append the compact search query, not the raw prompt: this entry goes + # straight to the search engine, which degrades sharply on prompt-shaped + # queries (issue #455). + queries.append(search_query) return queries, counter_callback @@ -630,7 +649,7 @@ def get_urls_from_queries_serper( ) response.raise_for_status() data = response.json() - organic = data.get("organic", []) + organic, _ = _shape_serper_sources(data, "live search") urls.extend(item["link"] for item in organic[:num]) except Exception as e: print(f"Error fetching URLs for query '{query}': {e}") @@ -1058,6 +1077,7 @@ def fetch_additional_information( # pylint: disable=too-many-statements,too-man client: "LLMClient", client_embedding: Optional["LLMClient"], prompt: str, + search_query: str, model: str, google_api_key: Optional[str], google_engine_id: Optional[str], @@ -1077,6 +1097,7 @@ def fetch_additional_information( # pylint: disable=too-many-statements,too-man queries, counter_callback = multi_queries( client=client, prompt=prompt, + search_query=search_query, model=model, num_queries=num_queries, counter_callback=counter_callback, @@ -1086,7 +1107,7 @@ def fetch_additional_information( # pylint: disable=too-many-statements,too-man print(f"Queries: {queries}") except Exception as e: print(f"Error generating queries: {e}") - queries = [prompt] + queries = [search_query] # get the top URLs for the queries if source_content is None: @@ -1160,7 +1181,12 @@ def fetch_additional_information( # pylint: disable=too-many-statements,too-man print(f"Split Docs: {len(split_docs)}") if len(split_docs) == 0: - raise ValueError("No valid documents found from the provided URLs") + # Empty retrieval (no URLs, no usable pages, or an empty cached + # capture): return an empty information block so run() can deliver + # the flagged null prediction instead of an unparseable exception + # string (issue #455). + print("No valid documents found; returning empty additional information") + return "", raw_source_content, queries, counter_callback if len(split_docs) > MAX_NR_DOCS: # truncate the split_docs to the first MAX_NR_DOCS documents @@ -1206,21 +1232,256 @@ def fetch_additional_information( # pylint: disable=too-many-statements,too-man return additional_information, raw_source_content, queries, counter_callback -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: - print(f"Error extracting question: {e}") - question = prompt +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc - return question + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + num_urls: int, + num_queries: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param num_urls: the num_urls recorded in used_params. + :param num_queries: the num_queries recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped search ("empty query") stays + distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[prediction-request-reasoning-v1] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "num_urls": num_urls, + "num_queries": num_queries, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params @with_key_rotation -def run( # pylint: disable=too-many-statements +def run( # pylint: disable=too-many-statements,too-many-locals **kwargs: Any, ) -> Union[MaxCostResponse, MechResponse]: """Run the task""" @@ -1250,7 +1511,30 @@ def run( # pylint: disable=too-many-statements llm_client, embedding_client, ): - prompt = extract_question(kwargs["prompt"]) + prompt = kwargs["prompt"] + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[prediction-request-reasoning-v1] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[prediction-request-reasoning-v1] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[prediction-request-reasoning-v1] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) max_tokens = kwargs.get("max_tokens", LLM_SETTINGS[model]["default_max_tokens"]) temperature = kwargs.get("temperature", LLM_SETTINGS[model]["temperature"]) num_urls = kwargs.get("num_urls", DEFAULT_NUM_URLS) @@ -1280,6 +1564,25 @@ def run( # pylint: disable=too-many-statements raise ValueError( f"Invalid source_content_mode: {source_content_mode!r}. Must be 'cleaned' or 'raw'." ) + if kwargs.get("source_content", None) is None and not any( + ch.isalnum() for ch in search_query + ): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # search and LLM calls and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + num_urls=num_urls, + num_queries=num_queries, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) ( additional_information, source_content, @@ -1288,7 +1591,8 @@ def run( # pylint: disable=too-many-statements ) = fetch_additional_information( client=llm_client, client_embedding=embedding_client, - prompt=prompt, + prompt=question, + search_query=search_query, model=model, google_api_key=google_api_key, google_engine_id=google_engine_id, @@ -1303,9 +1607,31 @@ def run( # pylint: disable=too-many-statements max_tokens=max_tokens, ) + if not additional_information: + # Retrieval produced no usable documents on either branch -- + # return the flagged null instead of an unparseable exception + # string (issue #455). + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + num_urls=num_urls, + num_queries=num_queries, + captured_source_content=source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context=( + "cached replay" + if kwargs.get("source_content", None) is not None + else "live search" + ), + tier=tier, + scan_truncated=scan_truncated, + ) + # Reasoning prompt reasoning_prompt = REASONING_PROMPT.format( - USER_PROMPT=prompt, ADDITIONAL_INFOMATION=additional_information + USER_PROMPT=question, ADDITIONAL_INFOMATION=additional_information ) # Do reasoning @@ -1324,7 +1650,7 @@ def run( # pylint: disable=too-many-statements # Prediction prompt prediction_prompt = PREDICTION_PROMPT.format( - USER_INPUT=prompt, REASONING=reasoning + USER_INPUT=question, REASONING=reasoning ) # Make the prediction @@ -1342,6 +1668,8 @@ def run( # pylint: disable=too-many-statements "max_tokens": max_tokens, "num_urls": num_urls, "num_queries": num_queries, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = source_content diff --git a/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py b/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py index 8014fce83..d55ffdeb7 100644 --- a/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py +++ b/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py @@ -20,6 +20,7 @@ """Unit tests for prediction_request_reasoning: thread-safe client, offline tiktoken, and source_content.""" import inspect +import json from concurrent.futures import Future from pathlib import Path from types import SimpleNamespace @@ -37,7 +38,10 @@ do_reasoning_with_retry, extract_texts, fetch_additional_information, + get_urls_from_queries_serper, + multi_queries, multi_questions_response, + parse_prompt, run, ) @@ -267,6 +271,7 @@ def test_cleaned_mode_uses_text_directly( client=MagicMock(), client_embedding=MagicMock(), prompt="test", + search_query="test", model="gpt-4.1-2025-04-14", google_api_key=None, google_engine_id=None, @@ -311,6 +316,7 @@ def test_raw_mode_re_extracts( client=MagicMock(), client_embedding=MagicMock(), prompt="test", + search_query="test", model="gpt-4.1-2025-04-14", google_api_key=None, google_engine_id=None, @@ -323,23 +329,28 @@ def test_raw_mode_re_extracts( assert "http://example.com" in result @patch(f"{REASONING_MODULE}.multi_queries") - def test_empty_source_content_raises(self, mock_queries: MagicMock) -> None: - """Empty source_content raises ValueError (no valid documents).""" + def test_empty_source_content_returns_empty_information( + self, mock_queries: MagicMock + ) -> None: + """Empty source_content yields an empty information block (flagged-null path).""" source_content: dict = {"pages": {}, "pdfs": {}} mock_queries.return_value = (["test query"], None) - with pytest.raises(ValueError, match="No valid documents"): - fetch_additional_information( - client=MagicMock(), - client_embedding=MagicMock(), - prompt="test", - model="gpt-4.1-2025-04-14", - google_api_key=None, - google_engine_id=None, - serper_api_key=None, - search_provider="google", - source_content=source_content, - ) + result, raw_sc, _, _ = fetch_additional_information( + client=MagicMock(), + client_embedding=MagicMock(), + prompt="test", + search_query="test", + model="gpt-4.1-2025-04-14", + google_api_key=None, + google_engine_id=None, + serper_api_key=None, + search_provider="google", + source_content=source_content, + ) + + assert result == "" + assert raw_sc is source_content def _make_mock_api_keys(return_source_content: str = "false") -> MagicMock: @@ -627,3 +638,319 @@ def test_anthropic_tokenizer_error_falls_back(self) -> None: result = count_tokens("hello world", "claude-sonnet-4-6", client=mock_client) assert isinstance(result, int) assert result > 0 + + +# --------------------------------------------------------------------------- +# Free-text input contract (issue #455): parse_prompt + flagged-null guards. +# --------------------------------------------------------------------------- + +# Trader-template format prompt (regression: previous callers must still work) +TRADER_PROMPT = ( + 'Given the question "Will X happen?" and the `yes` answer criterion, ...' +) +# Free-text format prompt: the advertised contract (issue #455) +FREE_TEXT_PROMPT = "Will Alexander Isak join Liverpool before September 2 2025?" +# Long free-text prompt that would return empty search results if passed raw +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently transfer " + "to Liverpool FC before the end of the summer 2025 transfer window (September 2, " + "2025 23:59 UTC)? Resolution source: official club announcements or BBC Sport. " + "The market resolves YES if a permanent transfer (not a loan) is confirmed by " + "the resolution source before the deadline." +) + + +def _make_serper_api_keys() -> MagicMock: + """Create a mock api_keys object routed to the (patched) Serper provider.""" + services = { + "openai": "sk-test", + "google_api_key": None, + "google_engine_id": None, + "serperapi": "serper-test", + "search_provider": "serper", + "return_source_content": "false", + } + mock_keys = MagicMock() + mock_keys.__getitem__ = MagicMock(side_effect=lambda k: services[k]) + mock_keys.get = MagicMock( + side_effect=lambda k, default=None: services.get(k, default) + ) + mock_keys.max_retries = MagicMock( + return_value={"openai": 0, "anthropic": 0, "google_api_key": 0, "openrouter": 0} + ) + return mock_keys + + +def _mock_client_manager(mock_mgr: MagicMock) -> MagicMock: + """Wire an LLMClientManager mock to yield (llm, embedding) client mocks.""" + mock_llm = MagicMock() + mock_mgr.return_value.__enter__ = MagicMock(return_value=(mock_llm, MagicMock())) + mock_mgr.return_value.__exit__ = MagicMock(return_value=False) + return mock_llm + + +class TestParsePromptContract: + """parse_prompt(): trader-template parity + free-text clause derivation.""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = parse_prompt(TRADER_PROMPT) + assert tier == "template" + assert question == "Will X happen?" + assert query == question + + def test_free_text_llm_gets_full_prompt(self) -> None: + """Free-text input: the LLM question is the whole prompt.""" + question, _, _ = parse_prompt(FREE_TEXT_PROMPT) + assert question == FREE_TEXT_PROMPT + question, _, _ = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert question == LONG_FREE_TEXT_PROMPT + + def test_boilerplate_prefix_is_dropped_from_query(self) -> None: + """The query anchors at the market question, dropping instruction text.""" + _, query, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert tier == "clause" + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert ( + len(query) <= module._MAX_SEARCH_QUERY_LEN + ) # pylint: disable=protected-access + + def test_tier_is_reported(self) -> None: + """The tier tags template / clause / raw explicitly.""" + assert parse_prompt(TRADER_PROMPT)[2] == "template" + assert parse_prompt(FREE_TEXT_PROMPT)[2] == "clause" + assert parse_prompt("no question mark here at all")[2] == "raw" + + +class TestDegenerateShortCircuit: + """Degenerate prompts return the flagged null with ZERO search calls.""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{REASONING_MODULE}.LLMClientManager") + @patch(f"{REASONING_MODULE}.get_urls_from_queries") + @patch(f"{REASONING_MODULE}.get_urls_from_queries_serper") + def test_degenerate_prompt_short_circuits( + self, + mock_serper: MagicMock, + mock_google: MagicMock, + mock_mgr: MagicMock, + degenerate: str, + ) -> None: + """Prompts with no searchable content never reach a search provider.""" + _mock_client_manager(mock_mgr) + result = run( + tool="prediction-request-reasoning-v1", + model="gpt-4.1-2025-04-14", + prompt=degenerate, + api_keys=_make_mock_api_keys(), + ) + mock_serper.assert_not_called() + mock_google.assert_not_called() + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["p_no"] == 0.5 + assert parsed["confidence"] == 0.0 and parsed["info_utility"] == 0.0 + used_params = result[4] + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "empty query" + assert used_params["scan_truncated"] is False + + +class TestEmptyRetrievalFlaggedNull: + """Empty retrieval yields a parseable flagged null, not an error string.""" + + @patch(f"{REASONING_MODULE}.LLMClientManager") + @patch(f"{REASONING_MODULE}.get_urls_from_queries_serper", return_value=[]) + def test_zero_urls_returns_flagged_null_live_search( + self, mock_serper: MagicMock, mock_mgr: MagicMock + ) -> None: + """A live search with no usable documents records null_reason='live search'.""" + _mock_client_manager(mock_mgr) + result = run( + tool="prediction-request-reasoning-v1", + model="gpt-4.1-2025-04-14", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_serper_api_keys(), + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["confidence"] == 0.0 + used_params = result[4] + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "live search" + assert used_params["parse_tier"] == "clause" + + @patch(f"{REASONING_MODULE}.LLMClientManager") + def test_empty_cached_replay_returns_flagged_null( + self, mock_mgr: MagicMock + ) -> None: + """An empty cached capture records null_reason='cached replay'.""" + _mock_client_manager(mock_mgr) + result = run( + tool="prediction-request-reasoning-v1", + model="gpt-4.1-2025-04-14", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + source_content={"pages": {}, "pdfs": {}}, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["confidence"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "cached replay" + + +class TestQueryLeakFix: + """The compact query, not the raw prompt, reaches the search engine.""" + + def test_multi_queries_appends_search_query_not_prompt(self) -> None: + """The direct-search append carries search_query; the LLM sees the prompt.""" + client = MagicMock() + client.completions.return_value = MagicMock( + content="alpha\nbeta", + usage=MagicMock(prompt_tokens=1, completion_tokens=1), + ) + search_query = "Will Alexander Isak permanently transfer to Liverpool FC?" + queries, _ = multi_queries( + client=client, + prompt=LONG_FREE_TEXT_PROMPT, + search_query=search_query, + model="gpt-4.1-2025-04-14", + num_queries=2, + ) + assert queries[-1] == search_query + assert LONG_FREE_TEXT_PROMPT not in queries + sent = client.completions.call_args.kwargs["messages"][1]["content"] + assert LONG_FREE_TEXT_PROMPT in sent + + @patch( + f"{REASONING_MODULE}.parser_prediction_response", return_value='{"p_yes": 0.5}' + ) + @patch(f"{REASONING_MODULE}.do_reasoning_with_retry") + @patch(f"{REASONING_MODULE}.fetch_additional_information") + @patch(f"{REASONING_MODULE}.LLMClientManager") + def test_run_feeds_question_to_llm_and_query_to_fetch( + self, + mock_mgr: MagicMock, + mock_fetch: MagicMock, + mock_reasoning: MagicMock, + mock_parser: MagicMock, + ) -> None: + """Free text: the LLM slots get the whole prompt, the search the clause.""" + mock_llm = _mock_client_manager(mock_mgr) + mock_fetch.return_value = ("additional info", {"pages": {}}, ["q"], None) + mock_reasoning.return_value = ("reasoning result", None) + mock_llm.completions.return_value = MagicMock( + content="0.5", + usage=MagicMock(prompt_tokens=10, completion_tokens=5), + ) + result = run( + tool="prediction-request-reasoning-v1", + model="gpt-4.1-2025-04-14", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + ) + fetch_kwargs = mock_fetch.call_args.kwargs + assert fetch_kwargs["prompt"] == LONG_FREE_TEXT_PROMPT + assert fetch_kwargs["search_query"].startswith("Will Alexander Isak") + assert fetch_kwargs["search_query"] != LONG_FREE_TEXT_PROMPT + # LLM-input parity: criteria the query drops still reach the LLM prompts + assert "official club announcements or BBC Sport" in result[1] + assert result[4]["parse_tier"] == "clause" + assert result[4]["scan_truncated"] is False + + @patch( + f"{REASONING_MODULE}.parser_prediction_response", return_value='{"p_yes": 0.5}' + ) + @patch(f"{REASONING_MODULE}.do_reasoning_with_retry") + @patch(f"{REASONING_MODULE}.fetch_additional_information") + @patch(f"{REASONING_MODULE}.LLMClientManager") + def test_trader_template_run_parity( + self, + mock_mgr: MagicMock, + mock_fetch: MagicMock, + mock_reasoning: MagicMock, + mock_parser: MagicMock, + ) -> None: + """Trader template: LLM question and search query stay the bare title.""" + mock_llm = _mock_client_manager(mock_mgr) + mock_fetch.return_value = ("additional info", {"pages": {}}, ["q"], None) + mock_reasoning.return_value = ("reasoning result", None) + mock_llm.completions.return_value = MagicMock( + content="0.5", + usage=MagicMock(prompt_tokens=10, completion_tokens=5), + ) + result = run( + tool="prediction-request-reasoning-v1", + model="gpt-4.1-2025-04-14", + prompt=TRADER_PROMPT, + api_keys=_make_mock_api_keys(), + ) + fetch_kwargs = mock_fetch.call_args.kwargs + assert fetch_kwargs["prompt"] == "Will X happen?" + assert fetch_kwargs["search_query"] == "Will X happen?" + # exactly the old extract_question behavior: bare question, no template + assert "Will X happen?" in result[1] + assert "`yes` answer criterion" not in result[1] + assert result[4]["parse_tier"] == "template" + + @patch( + f"{REASONING_MODULE}.parser_prediction_response", return_value='{"p_yes": 0.5}' + ) + @patch(f"{REASONING_MODULE}.do_reasoning_with_retry") + @patch(f"{REASONING_MODULE}.fetch_additional_information") + @patch(f"{REASONING_MODULE}.LLMClientManager") + def test_long_template_prompt_not_marked_truncated( + self, + mock_mgr: MagicMock, + mock_fetch: MagicMock, + mock_reasoning: MagicMock, + mock_parser: MagicMock, + ) -> None: + """Template past the window is NOT flagged: question precedes the scan.""" + mock_llm = _mock_client_manager(mock_mgr) + mock_fetch.return_value = ("additional info", {"pages": {}}, ["q"], None) + mock_reasoning.return_value = ("reasoning result", None) + mock_llm.completions.return_value = MagicMock( + content="0.5", + usage=MagicMock(prompt_tokens=10, completion_tokens=5), + ) + prompt = TRADER_PROMPT + " filler" * ( + module._MAX_SCAN_CHARS // 3 + ) # pylint: disable=protected-access + assert len(prompt) > module._MAX_SCAN_CHARS # pylint: disable=protected-access + result = run( + tool="prediction-request-reasoning-v1", + model="gpt-4.1-2025-04-14", + prompt=prompt, + api_keys=_make_mock_api_keys(), + ) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + +class TestSerperShapeGuard: + """Serper bodies are validated with the typed shape helper.""" + + @patch(f"{REASONING_MODULE}.requests.request") + def test_malformed_serper_body_is_skipped_not_crashed( + self, mock_request: MagicMock + ) -> None: + """A 200 body without the organic key is skipped for that query.""" + mock_request.return_value = MagicMock( + status_code=200, json=lambda: {"message": "quota exceeded"} + ) + urls = get_urls_from_queries_serper(["q1"], api_key="k", num=3) + assert urls == [] # pylint: disable=use-implicit-booleaness-not-comparison + + @patch(f"{REASONING_MODULE}.requests.request") + def test_valid_serper_body_yields_links(self, mock_request: MagicMock) -> None: + """A well-formed organic list still yields its links.""" + mock_request.return_value = MagicMock( + status_code=200, + json=lambda: { + "organic": [ + {"title": "T", "link": "https://example.test", "snippet": "S"} + ] + }, + ) + urls = get_urls_from_queries_serper(["q1"], api_key="k", num=3) + assert urls == ["https://example.test"] diff --git a/packages/packages.json b/packages/packages.json index 77686f975..2c0e49399 100644 --- a/packages/packages.json +++ b/packages/packages.json @@ -2,8 +2,8 @@ "dev": { "custom/dvilela/corcel_request/0.1.0": "bafybeic74xc32orfiffumv3cxe7o4vktzobk6gszjecuhto4or7k5ppi5y", "custom/dvilela/gemini_prediction/0.1.0": "bafybeifpoil2wjh6hr7sqoomlilhsyt3ktpy6tjzmisrlopxcjbkr3hhfq", - "custom/napthaai/prediction_request_rag_v1/0.1.0": "bafybeiesnbru6o3ochjn6s72iumcwmq2klgb3doa5pmleyt6psa52kxmjm", - "custom/napthaai/prediction_request_reasoning_v1/0.1.0": "bafybeigpzvqh2gbcqurqaefoopqpgmqu3dpy6p36l42pvfyx2iyy2ebvka", + "custom/napthaai/prediction_request_rag_v1/0.1.0": "bafybeierk5iyyf3tlpjqtxi5unawhih2sd2pwkp7owaxwszrxvatecdbdy", + "custom/napthaai/prediction_request_reasoning_v1/0.1.0": "bafybeicr7zpvpbg6kktaqi4shxpyok4gboqdbuxyd25zcsof6yljlffaie", "custom/napthaai/prediction_url_cot_v1/0.1.0": "bafybeife4x6syyji4jtfbg2jmxec7b2bv3z6d4lkvj6a7da5nhelrunyja", "custom/napthaai/resolve_market_reasoning/0.1.0": "bafybeidji6or6kpmawho64tc7qetinykhrklvv2gmnqfcd6ejmg43j3i7q", "custom/nickcom007/prediction_request_sme/0.1.0": "bafybeifsk4og24t22agcrj7mm6fz3wxwfbdg554yjhc6odpxal5ofg22ti", @@ -11,25 +11,25 @@ "custom/valory/factual_research_v1/0.1.0": "bafybeicg52wve2ytam3eisks7gumvrswm2v6bkuafburdlmh6d553nrvee", "custom/valory/factual_research_v2/0.1.0": "bafybeidgo3jesmvuwa64ooij7mywk7hf7ls5len2sf7cbcjwt3t6r5u6qm", "custom/valory/factual_research_v3/0.1.0": "bafybeiboyjc2u42lcsi7er6hywwdz27ae6r57y3jnqck5nk7tj4qosoyqa", - "custom/valory/finetuned_prediction/0.1.0": "bafybeiclpkn4sqvri7k5aiklt5fm6hnt3tgo4qxtrkzxe4fwzfs2plgbk4", + "custom/valory/finetuned_prediction/0.1.0": "bafybeifyhwavsm72okmmibc4yhxnlys5gcouhiydvathagpwnl5sgevsra", "custom/valory/prediction_langchain/0.1.0": "bafybeieirig3irwmfubxjoxcujiwxxb7knl3c3amhqhjy7bxvo2tdvybpm", "custom/valory/prediction_request_v1/0.1.0": "bafybeie23dpatwki6odlv7nnlwibuq6jbdonsmu5mszsgitlwghzpiesam", "custom/valory/prepare_tx/0.1.0": "bafybeidpvkwtn5m5yjp5mr2tn6f52btmlki32syahvkkuvlw3oq5eee3di", "custom/valory/propose_question/0.1.0": "bafybeif2dmjftef7pnixwnutn42tdaxbbr77anl5ixsv6kwsxnk7sm6fni", "custom/valory/resolve_market/0.1.0": "bafybeiaozazbggoglwrnfyn5v2qebyn4y4jpaxxbdodi3bd5eyieo4a55i", "custom/valory/resolve_market_jury/0.1.0": "bafybeigdv6zbbuf2flnti7hfwbotagnqz43g76v2jfhcuk4ancucxmg5gi", - "custom/valory/superforcaster/0.1.0": "bafybeidkraklf5c6rkv6g7y7lqj2woy4j5u5dmupjez6mtjhdek3lzbmki", - "custom/valory/superforcaster_calibrated_full_search/0.1.0": "bafybeialuhbtfumvlbdtzimfn7wdacqwccn2ny5cbhi4tkri6iih6pfgxe", - "custom/valory/superforcaster_full_search/0.1.0": "bafybeihba6ch6g7gndfwv75dtpfrpnvg4dw6wq6lxexch5h3n3vjafuxsm", + "custom/valory/superforcaster/0.1.0": "bafybeiggcearjnrowwmfucywlcytjth3g7fmm6f67j7unhaarvgqanm4q4", + "custom/valory/superforcaster_calibrated_full_search/0.1.0": "bafybeiblyiqyy3iae4wknmnty3sha4behm5eown4t34qdbuo4kjiie5hvm", + "custom/valory/superforcaster_full_search/0.1.0": "bafybeih7bm7zzy6e4ekkdupnuty4lhzum6tot6umum73j55mvp65yxv4ha", "custom/valory/superforcaster_market_aware/0.1.0": "bafybeiecvkdx2zjezc5aslnpg4nbw25rsbrc7xtpywbkjv3d47tdvskhgq", - "custom/valory/superforcaster_polymarket_v1/0.1.0": "bafybeiawfhy3w5aclmxm5zxebwd5l7paseqlgmihwbqfyhcdpn2kccgtrq", - "custom/valory/superforcaster_polymarket_v2/0.1.0": "bafybeigopot226az3ida2rwlamlr2isi4i2m3nk36dtzug3w3jvnk673uy", - "custom/valory/superforcaster_polymarket_v3/0.1.0": "bafybeiblm5te562opborjns6jngeqzgldklzegsm7csfgpzdps7y4mfpnu", + "custom/valory/superforcaster_polymarket_v1/0.1.0": "bafybeihaoz4gaiuoysiazfjxdn5uu3ybtirzdewuzzi35ixstwoa6ahi3u", + "custom/valory/superforcaster_polymarket_v2/0.1.0": "bafybeihdtol5qe6duatttwlsfvrxhpwgla2vy6hg2jnuyxu6x2niwhonsy", + "custom/valory/superforcaster_polymarket_v3/0.1.0": "bafybeibu3t7t6ea2pq4aiirxaz7srmltwihh4quxek35ctd4d3ra46wqzm", "custom/valory/superforcaster_polymarket_v4/0.1.0": "bafybeiefu5cetnebkr2yza6yn6la2ldkwvn2pjlgcmew2e6fblnyjh5roq", "custom/victorpolisetty/dalle_request/0.1.0": "bafybeiadatvfc6opcsmhpxxzmsqpwgbckblciqypx2iwe5uwove6nmki3e", "custom/victorpolisetty/gemini_request/0.1.0": "bafybeiamjk5mfycjqkstkzymggako2jt7yjros2zx4l54zyuei2xkj4gqu", - "agent/valory/mech_predict/0.1.0": "bafybeigyc7wnb6hko7rc7y2kwnvnckjsck4uffhlmdclrvpyp2jww7scrm", - "service/valory/mech_predict/0.1.0": "bafybeiclu26v6gwoqnme7py7wpdypsrusukqp4v3adpd4o7kpbw6kv32ja" + "agent/valory/mech_predict/0.1.0": "bafybeibafk2js4wyidwhjpdtddcjfsitbusnfnof6pcdxkxuggmdjmgary", + "service/valory/mech_predict/0.1.0": "bafybeiceslxfai4qdbire7natjjqbrlg4wclyye52elsopvg3sit5sb4vm" }, "third_party": { "protocol/open_aea/signing/1.0.0": "bafybeifsjmldwyki3beqyvdt5lzenrg6wyrqaar5plc5rpnvtc4zlentye", diff --git a/packages/valory/agents/mech_predict/aea-config.yaml b/packages/valory/agents/mech_predict/aea-config.yaml index b73a6dbb4..744022e5a 100644 --- a/packages/valory/agents/mech_predict/aea-config.yaml +++ b/packages/valory/agents/mech_predict/aea-config.yaml @@ -56,17 +56,17 @@ customs: - valory/resolve_market_jury:0.1.0:bafybeigdv6zbbuf2flnti7hfwbotagnqz43g76v2jfhcuk4ancucxmg5gi - valory/prediction_request_v1:0.1.0:bafybeie23dpatwki6odlv7nnlwibuq6jbdonsmu5mszsgitlwghzpiesam - napthaai/resolve_market_reasoning:0.1.0:bafybeidji6or6kpmawho64tc7qetinykhrklvv2gmnqfcd6ejmg43j3i7q -- napthaai/prediction_request_rag_v1:0.1.0:bafybeiesnbru6o3ochjn6s72iumcwmq2klgb3doa5pmleyt6psa52kxmjm -- napthaai/prediction_request_reasoning_v1:0.1.0:bafybeigpzvqh2gbcqurqaefoopqpgmqu3dpy6p36l42pvfyx2iyy2ebvka +- napthaai/prediction_request_rag_v1:0.1.0:bafybeierk5iyyf3tlpjqtxi5unawhih2sd2pwkp7owaxwszrxvatecdbdy +- napthaai/prediction_request_reasoning_v1:0.1.0:bafybeicr7zpvpbg6kktaqi4shxpyok4gboqdbuxyd25zcsof6yljlffaie - valory/prepare_tx:0.1.0:bafybeidpvkwtn5m5yjp5mr2tn6f52btmlki32syahvkkuvlw3oq5eee3di - napthaai/prediction_url_cot_v1:0.1.0:bafybeife4x6syyji4jtfbg2jmxec7b2bv3z6d4lkvj6a7da5nhelrunyja - valory/prediction_langchain:0.1.0:bafybeieirig3irwmfubxjoxcujiwxxb7knl3c3amhqhjy7bxvo2tdvybpm - victorpolisetty/gemini_request:0.1.0:bafybeiamjk5mfycjqkstkzymggako2jt7yjros2zx4l54zyuei2xkj4gqu - victorpolisetty/dalle_request:0.1.0:bafybeiadatvfc6opcsmhpxxzmsqpwgbckblciqypx2iwe5uwove6nmki3e -- valory/superforcaster:0.1.0:bafybeidkraklf5c6rkv6g7y7lqj2woy4j5u5dmupjez6mtjhdek3lzbmki -- valory/superforcaster_polymarket_v1:0.1.0:bafybeiawfhy3w5aclmxm5zxebwd5l7paseqlgmihwbqfyhcdpn2kccgtrq -- valory/superforcaster_polymarket_v2:0.1.0:bafybeigopot226az3ida2rwlamlr2isi4i2m3nk36dtzug3w3jvnk673uy -- valory/superforcaster_polymarket_v3:0.1.0:bafybeiblm5te562opborjns6jngeqzgldklzegsm7csfgpzdps7y4mfpnu +- valory/superforcaster:0.1.0:bafybeiggcearjnrowwmfucywlcytjth3g7fmm6f67j7unhaarvgqanm4q4 +- valory/superforcaster_polymarket_v1:0.1.0:bafybeihaoz4gaiuoysiazfjxdn5uu3ybtirzdewuzzi35ixstwoa6ahi3u +- valory/superforcaster_polymarket_v2:0.1.0:bafybeihdtol5qe6duatttwlsfvrxhpwgla2vy6hg2jnuyxu6x2niwhonsy +- valory/superforcaster_polymarket_v3:0.1.0:bafybeibu3t7t6ea2pq4aiirxaz7srmltwihh4quxek35ctd4d3ra46wqzm - valory/superforcaster_polymarket_v4:0.1.0:bafybeiefu5cetnebkr2yza6yn6la2ldkwvn2pjlgcmew2e6fblnyjh5roq - dvilela/corcel_request:0.1.0:bafybeic74xc32orfiffumv3cxe7o4vktzobk6gszjecuhto4or7k5ppi5y - dvilela/gemini_prediction:0.1.0:bafybeifpoil2wjh6hr7sqoomlilhsyt3ktpy6tjzmisrlopxcjbkr3hhfq @@ -74,8 +74,8 @@ customs: - valory/factual_research:0.1.0:bafybeidcxmqemvzm7cz2ydyjsmu53zflsqpcj2gpco5kwn3ldvf22rqx4i - valory/factual_research_v3:0.1.0:bafybeiboyjc2u42lcsi7er6hywwdz27ae6r57y3jnqck5nk7tj4qosoyqa - valory/propose_question:0.1.0:bafybeif2dmjftef7pnixwnutn42tdaxbbr77anl5ixsv6kwsxnk7sm6fni -- valory/superforcaster_full_search:0.1.0:bafybeihba6ch6g7gndfwv75dtpfrpnvg4dw6wq6lxexch5h3n3vjafuxsm -- valory/superforcaster_calibrated_full_search:0.1.0:bafybeialuhbtfumvlbdtzimfn7wdacqwccn2ny5cbhi4tkri6iih6pfgxe +- valory/superforcaster_full_search:0.1.0:bafybeih7bm7zzy6e4ekkdupnuty4lhzum6tot6umum73j55mvp65yxv4ha +- valory/superforcaster_calibrated_full_search:0.1.0:bafybeiblyiqyy3iae4wknmnty3sha4behm5eown4t34qdbuo4kjiie5hvm - valory/superforcaster_market_aware:0.1.0:bafybeiecvkdx2zjezc5aslnpg4nbw25rsbrc7xtpywbkjv3d47tdvskhgq default_ledger: ethereum required_ledgers: diff --git a/packages/valory/customs/finetuned_prediction/component.yaml b/packages/valory/customs/finetuned_prediction/component.yaml index 85c7ecb7f..06994053f 100644 --- a/packages/valory/customs/finetuned_prediction/component.yaml +++ b/packages/valory/customs/finetuned_prediction/component.yaml @@ -8,9 +8,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeid3gdbuetsikbutcl4i32avmicucmmrqmtnwn7gw454uizs5322se - finetuned_prediction.py: bafybeige7fgqpcn2qzedk5y2sp35gr32rscpobl4blnud2huo3mf3cfkku + finetuned_prediction.py: bafybeicwovt6uik2ahvzu5j4w7j6k7k5gdrrrrp5yushhykgxsh77bok7u tests/__init__.py: bafybeieut53j2sz53firzcldhpqeh2ca6rdopnywd7dnyzis4hvsprfs3m - tests/test_finetuned_prediction.py: bafybeidxnrj2uxwpkkdab2zp2s2qv4i6ifftlwoyy3yuirjv64yo7xcr3q + tests/test_finetuned_prediction.py: bafybeiemghiiukz2upy24ihdcj2xyffewrvs66wwmg6jpd6c4l5dm7rgg4 fingerprint_ignore_patterns: [] entry_point: finetuned_prediction.py callable: run diff --git a/packages/valory/customs/finetuned_prediction/finetuned_prediction.py b/packages/valory/customs/finetuned_prediction/finetuned_prediction.py index 15cfacc67..75662fb9f 100644 --- a/packages/valory/customs/finetuned_prediction/finetuned_prediction.py +++ b/packages/valory/customs/finetuned_prediction/finetuned_prediction.py @@ -71,7 +71,17 @@ import re import time from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import openai import requests @@ -90,6 +100,9 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 # Three modes, each a fixed vLLM served-model name. The tool NAME is the only # selector — `predict-base` calls the base model, `predict-fine-tuned` calls the @@ -605,28 +618,209 @@ def format_sources_data(organic_data: Any, misc_data: Any) -> str: return sources -def extract_question(prompt: str) -> str: - """Extract the market question from the mech prompt via regex.""" - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - return re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: # noqa: BLE001 — fall back to the whole prompt - print(f"Error extracting question: {e}") - return prompt +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc -def gather_sources(question: str, serper_api_key: str) -> str: - """Run the question through Serper and format the top results. +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. - Fails (raises, so with_key_rotation returns the error for the mech to - handle) on a Serper request error OR when Serper returns no usable results: - the model was trained only on research-backed prompts, so an empty - `` block is out-of-distribution — we fail the prediction with an - explanation rather than forecast on zero web context. + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) - :param question: the market question to search. + +def gather_sources(question: str, serper_api_key: str) -> Optional[str]: + """Run the search query through Serper and format the top results. + + Fails (raises, so with_key_rotation returns the error for the mech to + handle) on a Serper request error or a malformed response body (the typed + ValueError from _shape_serper_sources). A genuine zero-hit (organic AND + peopleAlsoAsk both empty) returns None instead so run() can deliver the + flagged null prediction (issue #455): the model was trained only on + research-backed prompts, so an empty `` block is + out-of-distribution and must not be forecast on. + + :param question: the search query to send to Serper. :param serper_api_key: the Serper API key. - :return: the formatted `` evidence block. + :return: the formatted `` evidence block, or None on zero hits. + :raises RuntimeError: if the Serper request itself fails. """ try: response = fetch_additional_sources(question, serper_api_key) @@ -634,16 +828,65 @@ def gather_sources(question: str, serper_api_key: str) -> str: except Exception as exc: # noqa: BLE001 — surface as an explanatory failure raise RuntimeError(f"Web search (Serper) request failed: {exc}") from exc - organic = data.get("organic", [])[:MAX_SOURCES] - misc = data.get("peopleAlsoAsk", []) + organic, misc = _shape_serper_sources(data, "live search") if not organic and not misc: - raise RuntimeError( - "Web search (Serper) returned no results; this model requires web " - "context and cannot forecast without it." - ) + return None return format_sources_data(organic, misc) +def _flagged_null_result( + *, + tool: str, + model: str, + temperature: float, + max_tokens: int, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so the strict trader consumer still parses it (issue #455). + The on-chain JSON carries only the four standard fields; the explicit + marker for requesters lives in used_params["empty_retrieval"] (off-chain + metadata.params), matching superforcaster-polymarket-v4. + + :param tool: the tool name recorded in used_params. + :param model: the served-model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[finetuned-prediction] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "tool": tool, + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + return null_result, "", None, counter_callback, used_params + + # --------------------------------------------------------------------------- # Prompt assembly # --------------------------------------------------------------------------- @@ -729,12 +972,62 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: temperature = float(kwargs.get("temperature", DEFAULT_SETTINGS["temperature"])) max_tokens = int(kwargs.get("max_tokens", DEFAULT_SETTINGS["max_tokens"])) - # Reproduce the production pipeline: the tool receives the bare-question - # prompt, pulls the question, runs web search, and builds the - # forecaster prompt the model trained on. + # Reproduce the production pipeline: the tool splits the request prompt + # into the LLM question and the Serper search query (issue #455), runs web + # search, and builds the forecaster prompt the model trained + # on. + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[finetuned-prediction] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[finetuned-prediction] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[finetuned-prediction] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) + + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + tool=tool, + model=model, + temperature=temperature, + max_tokens=max_tokens, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) + serper_api_key = api_keys["serperapi"] - question = extract_question(prompt) - sources = gather_sources(question, serper_api_key) + sources = gather_sources(search_query, serper_api_key) + if sources is None: + return _flagged_null_result( + tool=tool, + model=model, + temperature=temperature, + max_tokens=max_tokens, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) today = date.today().strftime(DATE_FORMAT) content = build_forecaster_prompt(question, today, sources) @@ -762,6 +1055,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } return result, completion, None, counter_callback, used_params diff --git a/packages/valory/customs/finetuned_prediction/tests/test_finetuned_prediction.py b/packages/valory/customs/finetuned_prediction/tests/test_finetuned_prediction.py index a67294c68..786cac58a 100644 --- a/packages/valory/customs/finetuned_prediction/tests/test_finetuned_prediction.py +++ b/packages/valory/customs/finetuned_prediction/tests/test_finetuned_prediction.py @@ -38,6 +38,7 @@ canonical_prediction, gather_sources, parse_p_yes, + parse_prompt, resolve_model, run, with_key_rotation, @@ -260,6 +261,10 @@ def test_run_fine_tuned_mode_calls_its_model_and_returns_canonical_json() -> Non assert used_params["model"] == MODEL_BY_TOOL[TOOL_FINE_TUNED] assert gen.call_args.kwargs["model"] == MODEL_BY_TOOL[TOOL_FINE_TUNED] + # The trader-template path records its tier and is never scan-truncated. + assert used_params["parse_tier"] == "template" + assert used_params["scan_truncated"] is False + # The question is extracted from the bare prompt and web-searched, then # embedded in the forecaster prompt as a single user message. gather.assert_called_once_with("Will X happen?", "serp-key") @@ -398,7 +403,7 @@ def test_vllm_client_passes_base_url() -> None: # --------------------------------------------------------------------------- -# gather_sources — fail closed when there is no web context (OOD for the model) +# gather_sources -- no web context (OOD for the model) yields None / an error # --------------------------------------------------------------------------- @@ -417,6 +422,7 @@ def test_gather_sources_formats_results() -> None: return_value=_serper_response(payload), ): out = gather_sources("Will X happen?", "serp-key") + assert out is not None assert "Organic Results" in out and "T" in out @@ -429,26 +435,151 @@ def test_gather_sources_raises_on_serper_request_failure() -> None: gather_sources("Will X happen?", "serp-key") -def test_gather_sources_raises_on_zero_results() -> None: - """Zero usable results becomes an explanatory failure (model is OOD without sources).""" +def test_gather_sources_returns_none_on_zero_results() -> None: + """Zero usable results returns None so run() can deliver the flagged null.""" payload: dict[str, list] = {"organic": [], "peopleAlsoAsk": []} with patch( f"{MODULE_PATH}.fetch_additional_sources", return_value=_serper_response(payload), ): - with pytest.raises(RuntimeError, match="no results"): - gather_sources("Will X happen?", "serp-key") + assert gather_sources("Will X happen?", "serp-key") is None + + +# --------------------------------------------------------------------------- +# parse_prompt + empty-retrieval guard (issue #455 port) +# --------------------------------------------------------------------------- + + +def test_parse_prompt_trader_template_parity() -> None: + """Trader-template path: the extracted title serves as BOTH values.""" + question, query, tier = parse_prompt(_bare_prompt("Will X happen by 2026?")) + assert tier == "template" + # LLM-input parity with the old extract_question behavior: the question + # fed to the LLM and the search query are both the extracted title. + assert question == "Will X happen by 2026?" + assert query == question + + +def test_parse_prompt_free_text_clause_derivation() -> None: + """A boilerplate lead-in anchors the market question as the search query.""" + prompt = ( + "Please predict the following market: Will Alexander Isak permanently " + "transfer to Liverpool FC before September 2, 2025? Resolution source: " + "official club announcements or BBC Sport." + ) + question, query, tier = parse_prompt(prompt) + assert tier == "clause" + # The LLM question is the WHOLE prompt; only the search query is derived. + assert question == prompt + assert query.startswith("Will Alexander Isak") + assert query.endswith("2025?") + + +@pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) +def test_degenerate_prompt_short_circuits_before_search(degenerate: str) -> None: + """Unsearchable prompts return the flagged null with ZERO search calls.""" + keychain = FakeKeyChain({"finetuned": "EMPTY", "serperapi": "serp-key"}) + with ( + patch(f"{MODULE_PATH}.fetch_additional_sources") as fetch, + patch(f"{MODULE_PATH}.generate_prediction_with_retry") as gen, + patch(f"{MODULE_PATH}.VLLMClientManager"), + ): + out = run(tool=TOOL_BASE, prompt=degenerate, api_keys=keychain) + fetch.assert_not_called() + gen.assert_not_called() + result, _, tx, _, used_params, _ = out + assert json.loads(result) == { + "p_yes": 0.5, + "p_no": 0.5, + "confidence": 0.0, + "info_utility": 0.0, + } + assert tx is None + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "empty query" + assert used_params["scan_truncated"] is False -def test_run_fails_with_explanation_when_serper_returns_nothing() -> None: - """End-to-end: empty Serper results surface the explanatory message as the result.""" +def test_both_empty_retrieval_returns_flagged_null_live_search() -> None: + """Organic AND peopleAlsoAsk both empty -> flagged null, 'live search'.""" keychain = FakeKeyChain({"finetuned": "EMPTY", "serperapi": "serp-key"}) payload: dict[str, list] = {"organic": [], "peopleAlsoAsk": []} with ( + patch( + f"{MODULE_PATH}.fetch_additional_sources", + return_value=_serper_response(payload), + ), + patch(f"{MODULE_PATH}.generate_prediction_with_retry") as gen, + patch(f"{MODULE_PATH}.VLLMClientManager"), + ): + out = run( + tool=TOOL_BASE, + prompt=_bare_prompt("Will X happen?"), + api_keys=keychain, + ) + gen.assert_not_called() + result, _, _, _, used_params, _ = out + assert json.loads(result)["p_yes"] == 0.5 + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "live search" + assert used_params["parse_tier"] == "template" + + +def test_template_past_scan_window_not_marked_truncated() -> None: + """A trader-template prompt longer than the window is NOT scan_truncated.""" + keychain = FakeKeyChain({"finetuned": "EMPTY", "serperapi": "serp-key"}) + scan_chars = module._MAX_SCAN_CHARS + prompt = _bare_prompt("Will X happen?") + " filler" * (scan_chars // 3) + assert len(prompt) > scan_chars + with ( + patch(f"{MODULE_PATH}.generate_prediction_with_retry") as gen, + patch(f"{MODULE_PATH}.VLLMClientManager"), + patch(f"{MODULE_PATH}.gather_sources", return_value="SRC"), + ): + gen.return_value = (WELL_FORMED, None) + out = run(tool=TOOL_BASE, prompt=prompt, api_keys=keychain) + used_params = out[4] + assert used_params["parse_tier"] == "template" + assert used_params["scan_truncated"] is False + + +def test_free_text_search_uses_derived_query_not_full_prompt() -> None: + """The Serper call gets the derived clause; the LLM gets the whole prompt.""" + keychain = FakeKeyChain({"finetuned": "EMPTY", "serperapi": "serp-key"}) + prompt = ( + "Please predict the following market: Will Alexander Isak permanently " + "transfer to Liverpool FC before September 2, 2025? Resolution source: " + "official club announcements or BBC Sport." + ) + payload = {"organic": [{"position": 1, "title": "T", "link": "L", "snippet": "S"}]} + with ( + patch(f"{MODULE_PATH}.generate_prediction_with_retry") as gen, patch(f"{MODULE_PATH}.VLLMClientManager"), patch( f"{MODULE_PATH}.fetch_additional_sources", return_value=_serper_response(payload), + ) as fetch, + ): + gen.return_value = (WELL_FORMED, None) + out = run(tool=TOOL_BASE, prompt=prompt, api_keys=keychain) + query = fetch.call_args.args[0] + assert query.startswith("Will Alexander Isak") + assert len(query) < len(prompt) + # The full prompt (resolution criteria included) reaches the forecaster + # template, not the compressed search query. + user_content = gen.call_args.kwargs["messages"][0]["content"] + assert "official club announcements or BBC Sport" in user_content + assert out[4]["parse_tier"] == "clause" + + +def test_malformed_serper_body_is_typed_error_not_flagged_null() -> None: + """organic: null surfaces as the shape ValueError, not a flagged null.""" + keychain = FakeKeyChain({"finetuned": "EMPTY", "serperapi": "serp-key"}) + with ( + patch(f"{MODULE_PATH}.VLLMClientManager"), + patch( + f"{MODULE_PATH}.fetch_additional_sources", + return_value=_serper_response({"organic": None, "peopleAlsoAsk": []}), ), ): out = run( @@ -456,5 +587,5 @@ def test_run_fails_with_explanation_when_serper_returns_nothing() -> None: prompt=_bare_prompt("Will X happen?"), api_keys=keychain, ) - # with_key_rotation converts the RuntimeError into an error result tuple. - assert "no results" in out[0] + # with_key_rotation converts the ValueError into an error result tuple. + assert "malformed 'organic'" in out[0] diff --git a/packages/valory/customs/superforcaster/component.yaml b/packages/valory/customs/superforcaster/component.yaml index 53420e546..dc8dd52dd 100644 --- a/packages/valory/customs/superforcaster/component.yaml +++ b/packages/valory/customs/superforcaster/component.yaml @@ -8,9 +8,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeifvbuxt54l5jsxextf6ru5yvimkfdg4gfkcsnmyvsyqcsgclg7vey - superforcaster.py: bafybeic7vbcjcq5jomciosjaez2nnktwsdlbf2ctakdrbh5nl7cgod5z6e + superforcaster.py: bafybeieigrs2mxxtai7vpalhrs353akbewcs7oahgtud6htljbsg2fdgzy tests/__init__.py: bafybeidegv7yfxpdytmvepvcqquzawanqjczrqdp2cesxxdx5vvdfmdgue - tests/test_superforcaster.py: bafybeiboaixfm4itoashxu5a4fshfe36plhnkk2tw3reznnoxz4yl7ovhm + tests/test_superforcaster.py: bafybeif5i27k7nll3bt6tu2lwc6t3ck5lynzkd54iktpamwzkyjwesr75i fingerprint_ignore_patterns: [] entry_point: superforcaster.py callable: run diff --git a/packages/valory/customs/superforcaster/superforcaster.py b/packages/valory/customs/superforcaster/superforcaster.py index 2dafad8e0..7302a4a39 100644 --- a/packages/valory/customs/superforcaster/superforcaster.py +++ b/packages/valory/customs/superforcaster/superforcaster.py @@ -23,7 +23,17 @@ import re import time from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import openai import requests @@ -41,6 +51,9 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 # --------------------------------------------------------------------------- @@ -397,7 +410,10 @@ def fetch_additional_sources(question: Any, serper_api_key: Any) -> requests.Res "Content-Type": "application/json", } - response = requests.request("POST", url, headers=headers, data=payload) + # timeout matches the fleet's other Serper callers (factual_research, + # superforcaster_calibrated_full_search); a hung connection must not block + # the task indefinitely. + response = requests.request("POST", url, headers=headers, data=payload, timeout=30) return response @@ -435,16 +451,247 @@ def format_sources_data(organic_data: Any, misc_data: Any) -> str: return sources -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: - print(f"Error extracting question: {e}") - question = prompt - return question +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc + + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). The on-chain JSON carries + only the four standard fields; the explicit marker for requesters lives + in used_params["empty_retrieval"] (off-chain metadata.params). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[superforcaster] {context}: empty retrieval" " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params @with_key_rotation @@ -492,19 +739,73 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: today = date.today() d = today.strftime("%d/%m/%Y") - question = extract_question(prompt) + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[superforcaster] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[superforcaster] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[superforcaster] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) if source_content is not None: print("Using provided source content (cached replay)...") captured_source_content = source_content serper_data = source_content.get("serper_response", source_content) - organic_data = serper_data.get("organic", [])[:MAX_SOURCES] - misc_data = serper_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources( + serper_data, "cached replay" + ) + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="cached replay", + tier=tier, + scan_truncated=scan_truncated, + ) sources = format_sources_data(organic_data, misc_data) else: + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) serper_api_key = kwargs["api_keys"]["serperapi"] print("Fetching additional sources...") - serper_response = fetch_additional_sources(question, serper_api_key) + serper_response = fetch_additional_sources(search_query, serper_api_key) + # Raise on a 4xx/5xx error body (credit / auth error) instead of + # calling .json() on it and feeding the model an empty + # block that looks like a healthy run (matches the fleet pattern). + serper_response.raise_for_status() sources_data = serper_response.json() # mode tag included for consistency across tools; content is identical # regardless of mode since Serper returns structured JSON, not HTML @@ -513,8 +814,19 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "serper_response": sources_data, } print(f"Additional sources fetched: {sources_data}") - organic_data = sources_data.get("organic", [])[:MAX_SOURCES] - misc_data = sources_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources(sources_data, "live search") + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) print("Formating sources...") sources = format_sources_data(organic_data, misc_data) @@ -564,6 +876,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = captured_source_content diff --git a/packages/valory/customs/superforcaster/tests/test_superforcaster.py b/packages/valory/customs/superforcaster/tests/test_superforcaster.py index 55a7d63b4..4eadccc2c 100644 --- a/packages/valory/customs/superforcaster/tests/test_superforcaster.py +++ b/packages/valory/customs/superforcaster/tests/test_superforcaster.py @@ -30,7 +30,10 @@ from packages.valory.customs.superforcaster.superforcaster import ( OpenAIClientManager, PredictionResult, + _MAX_SCAN_CHARS, + _MAX_SEARCH_QUERY_LEN, _parse_completion, + parse_prompt, run, ) @@ -312,3 +315,196 @@ def test_flag_off_no_source_content( used_params = result[4] assert "source_content" not in used_params + + +FREE_TEXT_PROMPT = "Will Alexander Isak join Liverpool before September 2 2025?" +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently transfer " + "to Liverpool FC before the end of the summer 2025 transfer window (September 2, " + "2025 23:59 UTC)? Resolution source: official club announcements or BBC Sport. " + "The market resolves YES if a permanent transfer (not a loan) is confirmed by " + "the resolution source before the deadline." +) + +EMPTY_SERPER_RESPONSE: dict = {"organic": [], "peopleAlsoAsk": []} + + +class TestParsePromptContract: + """parse_prompt() -> (question_for_llm, search_query, tier) -- issue #455.""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = parse_prompt(PREDICTION_PROMPT) + assert question == "Will X happen?" + assert query == question + assert tier == "template" + + def test_free_text_llm_gets_full_prompt(self) -> None: + """Free-text input: the LLM question is the whole prompt.""" + question, _, _ = parse_prompt(FREE_TEXT_PROMPT) + assert question == FREE_TEXT_PROMPT + question, _, _ = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert question == LONG_FREE_TEXT_PROMPT + + def test_boilerplate_prefix_is_dropped_from_query(self) -> None: + """The query anchors at the market question, dropping instruction text.""" + _, query, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert len(query) <= _MAX_SEARCH_QUERY_LEN + assert tier == "clause" + + +class TestEmptyRetrievalGuard: + """Degenerate prompts short-circuit; zero-hit retrieval returns a flagged null.""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_degenerate_prompt_short_circuits_before_serper( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock, degenerate: str + ) -> None: + """Prompts with no searchable content never reach Serper at all.""" + result = run( + tool="superforcaster", + model="gpt-4.1-2025-04-14", + prompt=degenerate, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_fetch.assert_not_called() + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 + assert parsed["confidence"] == 0.0 + used_params = result[4] + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "empty query" + assert used_params["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_both_empty_live_retrieval_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Organic AND peopleAlsoAsk empty -> flagged null, no LLM call.""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + mock_client = MagicMock() + mock_client_mgr.return_value.__enter__ = MagicMock(return_value=mock_client) + mock_client_mgr.return_value.__exit__ = MagicMock(return_value=False) + + result = run( + tool="superforcaster", + model="gpt-4.1-2025-04-14", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_client.beta.chat.completions.parse.assert_not_called() + assert json.loads(result[0])["p_yes"] == 0.5 + used_params = result[4] + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "live search" + assert used_params["parse_tier"] == "clause" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_both_empty_cached_replay_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """An empty cached capture takes the guard too, without a Serper call.""" + result = run( + tool="superforcaster", + model="gpt-4.1-2025-04-14", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + source_content={"serper_response": EMPTY_SERPER_RESPONSE}, + ) + mock_fetch.assert_not_called() + assert json.loads(result[0])["p_yes"] == 0.5 + used_params = result[4] + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "cached replay" + + +class TestScanWindowObservability: + """parse_tier / scan_truncated reach used_params on every path.""" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_long_template_prompt_is_not_marked_truncated( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Template past the window is NOT flagged: the match precedes the scan.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_client = MagicMock() + mock_client.beta.chat.completions.parse.return_value = _mock_parse_response() + mock_client_mgr.return_value.__enter__ = MagicMock(return_value=mock_client) + mock_client_mgr.return_value.__exit__ = MagicMock(return_value=False) + + prompt = PREDICTION_PROMPT + " filler" * (_MAX_SCAN_CHARS // 3) + assert len(prompt) > _MAX_SCAN_CHARS + result = run( + tool="superforcaster", + model="gpt-4.1-2025-04-14", + prompt=prompt, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + used_params = result[4] + assert used_params["parse_tier"] == "template" + assert used_params["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_trader_request_sends_extracted_question_to_serper( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """LLM-input parity pin: the trader path is byte-identical to before.""" + serper_resp = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_fetch.return_value = serper_resp + mock_client = MagicMock() + mock_client.beta.chat.completions.parse.return_value = _mock_parse_response() + mock_client_mgr.return_value.__enter__ = MagicMock(return_value=mock_client) + mock_client_mgr.return_value.__exit__ = MagicMock(return_value=False) + + run( + tool="superforcaster", + model="gpt-4.1-2025-04-14", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + # the HTTP-error guard must actually run on the happy path + serper_resp.raise_for_status.assert_called_once() + query_sent = mock_fetch.call_args[0][0] + assert query_sent == "Will X happen?" + # and the LLM prompt carries the bare question, not the full template + llm_prompt = mock_client.beta.chat.completions.parse.call_args[1]["messages"][ + 1 + ]["content"] + assert "Will X happen?" in llm_prompt + assert "`yes` option represented" not in llm_prompt + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_success_used_params_carry_parse_tier( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """A free-text success records parse_tier=clause, not truncated.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_client = MagicMock() + mock_client.beta.chat.completions.parse.return_value = _mock_parse_response() + mock_client_mgr.return_value.__enter__ = MagicMock(return_value=mock_client) + mock_client_mgr.return_value.__exit__ = MagicMock(return_value=False) + + result = run( + tool="superforcaster", + model="gpt-4.1-2025-04-14", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + used_params = result[4] + assert used_params["parse_tier"] == "clause" + assert used_params["scan_truncated"] is False diff --git a/packages/valory/customs/superforcaster_calibrated_full_search/component.yaml b/packages/valory/customs/superforcaster_calibrated_full_search/component.yaml index 6b2500e4d..f125ed127 100644 --- a/packages/valory/customs/superforcaster_calibrated_full_search/component.yaml +++ b/packages/valory/customs/superforcaster_calibrated_full_search/component.yaml @@ -13,9 +13,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeif65opovmxvdk2yg56jsgauiuhjppuzrk4zvvruzklko3nmk5tjfm - superforcaster_calibrated_full_search.py: bafybeihq5w6g5crqmb5uiixrrbftz35j36zc4vbew5rfeexigr47ikah7y + superforcaster_calibrated_full_search.py: bafybeigr5ihsof7jh7lrvw5lerm7vmf3v7ksmsroul2wqghtbr2r6g5thy tests/__init__.py: bafybeigjdbprffbvsm6xzzuuzlk7kme5pl7bfib5ce72h35vq4sajfu5hi - tests/test_superforcaster_calibrated_full_search.py: bafybeihfmhgvvympjumsx4yrd3cx6i3ajlsu2mh6bzymh2bernrfa6oucq + tests/test_superforcaster_calibrated_full_search.py: bafybeics74smgobcqwsz7e54nncfwue62gr2cqz235rd6n5lx6g2fwe734 fingerprint_ignore_patterns: [] entry_point: superforcaster_calibrated_full_search.py callable: run diff --git a/packages/valory/customs/superforcaster_calibrated_full_search/superforcaster_calibrated_full_search.py b/packages/valory/customs/superforcaster_calibrated_full_search/superforcaster_calibrated_full_search.py index f244977c7..98529f346 100644 --- a/packages/valory/customs/superforcaster_calibrated_full_search/superforcaster_calibrated_full_search.py +++ b/packages/valory/customs/superforcaster_calibrated_full_search/superforcaster_calibrated_full_search.py @@ -31,7 +31,17 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import openai import requests @@ -51,6 +61,9 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 # --------------------------------------------------------------------------- @@ -639,15 +652,250 @@ def _cap_evidence_block( return rendered -def extract_question(prompt: str) -> str: - """Use regexp to extract the question from the prompt.""" - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: # noqa: BLE001 - print(f"Error extracting question: {e}") - question = prompt - return question +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc + + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + Unlike the decorator's error JSON this is a VALID prediction + (p_yes = p_no = 0.5) with zero confidence and info_utility, so the strict + trader consumer still parses it (issue #455). The on-chain JSON carries + only the four standard fields; the explicit marker for requesters lives in + used_params["empty_retrieval"] (off-chain metadata.params), intended for + off-chain consumers (not yet wired up -- the benchmark scorer's + null-vs-forecast branch is a follow-up). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[superforcaster_calibrated_full_search] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params @with_key_rotation @@ -695,33 +943,104 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: today = date.today() d = today.strftime("%d/%m/%Y") - question = extract_question(prompt) + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[superforcaster_calibrated_full_search] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[superforcaster_calibrated_full_search] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[superforcaster_calibrated_full_search] Free-text prompt " + f"(tier={tier}); derived search query: {search_query!r}" + ) if source_content is not None: print("Using provided source content (cached replay)...") captured_source_content = source_content serper_data = source_content.get("serper_response", source_content) - organic_data = [ - dict(it) for it in serper_data.get("organic", [])[:MAX_SOURCES] - ] - misc_data = serper_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources( + serper_data, "cached replay" + ) + # Shallow-copy each organic item so attaching `content` does not + # mutate the caller's cached source_content payload. + organic_data = [dict(it) for it in organic_data] + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="cached replay", + tier=tier, + scan_truncated=scan_truncated, + ) cached_pages = source_content.get("pages", {}) cached_mode = source_content.get("mode", source_content_mode) _hydrate_organic_from_pages(organic_data, cached_pages, cached_mode) sources = _cap_evidence_block(organic_data, misc_data, model) else: + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) serper_api_key = kwargs["api_keys"]["serperapi"] print("Fetching additional sources...") - serper_response = fetch_additional_sources(question, serper_api_key) + # Use the compressed search_query instead of the full prompt so + # Serper returns organic results for free-text callers (issue #455). + serper_response = fetch_additional_sources(search_query, serper_api_key) # Surface HTTP errors with a real status code instead of crashing # .json() on a non-JSON 4xx/5xx body (matches the fleet pattern). serper_response.raise_for_status() sources_data = serper_response.json() print(f"Additional sources fetched: {sources_data}") - organic_data = [ - dict(it) for it in sources_data.get("organic", [])[:MAX_SOURCES] - ] - misc_data = sources_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources(sources_data, "live search") + # Shallow-copy organic items: _scrape_pages attaches `content`, + # and we don't want that leaking into the captured serper_response. + organic_data = [dict(it) for it in organic_data] + # Empty-retrieval guard: even a correct short query can fail + # (e.g. very niche or recent market). Placed BEFORE the scraping + # step so empty retrieval wastes no page fetches (issue #455). + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content={ + "mode": source_content_mode, + "serper_response": sources_data, + }, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) print("Scraping page content for top organic results...") captured_pages = _scrape_pages(organic_data, source_content_mode) print( @@ -790,6 +1109,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = captured_source_content diff --git a/packages/valory/customs/superforcaster_calibrated_full_search/tests/test_superforcaster_calibrated_full_search.py b/packages/valory/customs/superforcaster_calibrated_full_search/tests/test_superforcaster_calibrated_full_search.py index 2cb28d698..2f38c2a55 100644 --- a/packages/valory/customs/superforcaster_calibrated_full_search/tests/test_superforcaster_calibrated_full_search.py +++ b/packages/valory/customs/superforcaster_calibrated_full_search/tests/test_superforcaster_calibrated_full_search.py @@ -38,12 +38,15 @@ MAX_EVIDENCE_TOKENS, OpenAIClientManager, PredictionResult, + _MAX_SCAN_CHARS, + _MAX_SEARCH_QUERY_LEN, _cap_evidence_block, _fetch_page_content, _parse_completion, _scrape_pages, count_tokens, fetch_additional_sources, + parse_prompt, run, ) @@ -678,3 +681,251 @@ def test_max_cost_returns_float_not_wrapped_tuple(self) -> None: delivery_rate=0, ) assert result == 0.0123 + + +# --- issue #455: free-text contract (ported from superforcaster-polymarket-v4, +# FINAL implementation as merged in PRs #459 + #461) --- + +FREE_TEXT_PROMPT = "Will Alexander Isak join Liverpool before September 2 2025?" +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently " + "transfer to Liverpool FC before the end of the summer 2025 transfer " + "window (September 2, 2025 23:59 UTC)? Resolution source: official club " + "announcements or BBC Sport. The market resolves YES if a permanent " + "transfer (not a loan) is confirmed by the deadline." +) +EMPTY_SERPER_RESPONSE: dict = { + "searchParameters": {"q": "x", "type": "search"}, + "organic": [], + "peopleAlsoAsk": [], +} + + +class TestParsePrompt: + """Test parse_prompt -> ParsedPrompt(question, query, tier) (issue #455).""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = parse_prompt(PREDICTION_PROMPT) + assert question == "Will X happen?" + assert query == question + assert tier == "template" + + def test_free_text_llm_gets_full_prompt(self) -> None: + """Free-text input: the LLM question is the whole prompt.""" + question, _, _ = parse_prompt(FREE_TEXT_PROMPT) + assert question == FREE_TEXT_PROMPT + question, _, _ = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert question == LONG_FREE_TEXT_PROMPT + + def test_boilerplate_prefix_is_dropped_from_query(self) -> None: + """The query anchors at the question clause, dropping instruction text.""" + _, query, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert len(query) <= _MAX_SEARCH_QUERY_LEN + assert tier == "clause" + + def test_scan_window_bounds_candidate_search(self) -> None: + """The scan must not see past the window (deterministic cap killer). + + The only question clause sits beyond _MAX_SCAN_CHARS: a bounded scan + finds nothing (raw tier); removing the cap finds the clause and flips + the tier, failing this test with no reliance on wall-clock. + """ + prompt = "x" * (3 * _MAX_SCAN_CHARS) + " Will X happen by 2027?" + question, _, tier = parse_prompt(prompt) + assert tier == "raw" + assert question == prompt # the LLM still gets everything + + +class TestDegeneratePromptShortCircuit: + """Degenerate prompts never reach Serper; they return the flagged null.""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_degenerate_prompt_never_sends_an_empty_query( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock, degenerate: str + ) -> None: + """Prompts with no searchable content never reach Serper at all.""" + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=degenerate, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + mock_fetch.assert_not_called() + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["p_no"] == 0.5 + assert parsed["confidence"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "empty query" + assert result[4]["scan_truncated"] is False + + +class TestEmptyRetrievalGuard: + """Empty retrieval returns a flagged, tier-tagged null (issue #455).""" + + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_empty_live_retrieval_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Organic + PAA both empty -> null payload with markers, no LLM call.""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys("true"), + counter_callback=None, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["p_no"] == 0.5 + assert parsed["confidence"] == 0.0 + used_params = result[4] + assert used_params["empty_retrieval"] is True + assert used_params["null_reason"] == "live search" + assert used_params["parse_tier"] == "clause" + assert used_params["source_content"]["serper_response"] == ( + EMPTY_SERPER_RESPONSE + ) + + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_cached_replay_empty_retrieval_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """The cached-replay branch fires the same guard without a live fetch.""" + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + source_content={"serper_response": EMPTY_SERPER_RESPONSE}, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "cached replay" + mock_fetch.assert_not_called() + + @patch(f"{SFC_MODULE}._scrape_pages") + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_empty_retrieval_skips_scraping( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + mock_scrape: MagicMock, + ) -> None: + """The guard fires BEFORE the page-scrape step (no wasted fetches).""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + mock_scrape.assert_not_called() + + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_reshaped_serper_body_is_an_error_not_a_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """A 200 body without the organic key surfaces as an error null.""" + mock_fetch.return_value = MagicMock(json=lambda: {"message": "quota"}) + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] is None + assert "organic" in parsed["error"] + + +class TestLlmInputParityAndObservability: + """Trader-path parity pins + the parse-tier observability markers.""" + + @patch(f"{SFC_MODULE}._fetch_page_content", side_effect=_fake_fetch) + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_trader_request_sends_extracted_question_to_serper( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + _mock_page_fetch: MagicMock, + ) -> None: + """End-to-end pin: a trader request searches the bare question.""" + serper_resp = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_fetch.return_value = serper_resp + _stub_openai(mock_client_mgr) + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + serper_resp.raise_for_status.assert_called_once() + assert mock_fetch.call_args[0][0] == "Will X happen?" + # the LLM prompt (result[1]) carries the bare question, not the template + assert "Will X happen?" in result[1] + assert "and the `yes` option" not in result[1] + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + @patch(f"{SFC_MODULE}._fetch_page_content", side_effect=_fake_fetch) + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_free_text_searches_clause_but_llm_gets_full_prompt( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + _mock_page_fetch: MagicMock, + ) -> None: + """Free text: Serper gets the derived clause, the LLM the whole prompt.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + _stub_openai(mock_client_mgr) + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + assert mock_fetch.call_args[0][0].startswith("Will Alexander Isak") + assert "official club announcements or BBC Sport" in result[1] + assert result[4]["parse_tier"] == "clause" + + @patch(f"{SFC_MODULE}._fetch_page_content", side_effect=_fake_fetch) + @patch(f"{SFC_MODULE}.OpenAIClientManager") + @patch(f"{SFC_MODULE}.fetch_additional_sources") + def test_long_template_prompt_is_not_marked_truncated( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + _mock_page_fetch: MagicMock, + ) -> None: + """Template past the window is NOT flagged: question precedes the scan.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + _stub_openai(mock_client_mgr) + prompt = PREDICTION_PROMPT + " filler" * (_MAX_SCAN_CHARS // 3) + assert len(prompt) > _MAX_SCAN_CHARS + result = run( + tool="superforcaster_calibrated_full_search", + model="gpt-4o", + prompt=prompt, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False diff --git a/packages/valory/customs/superforcaster_full_search/component.yaml b/packages/valory/customs/superforcaster_full_search/component.yaml index c9b3242c8..00439c951 100644 --- a/packages/valory/customs/superforcaster_full_search/component.yaml +++ b/packages/valory/customs/superforcaster_full_search/component.yaml @@ -11,9 +11,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeid6bfykvdv4tejg2xfnpv3mrzzxlzuqfni5av4yfigz6l3qckffde - superforcaster_full_search.py: bafybeigydebqg7mahm3uap32qymfpdmrudvnwmwxulax2ky4mrh6t3owfe + superforcaster_full_search.py: bafybeidfgetrsypibet5zthpwvxm34imv26qhvm7xlpbk6bq3u34eicviy tests/__init__.py: bafybeidegv7yfxpdytmvepvcqquzawanqjczrqdp2cesxxdx5vvdfmdgue - tests/test_superforcaster_full_search.py: bafybeihromvcvabbnzkrl4neyj3r3j6fkdrggwfv57cwjtsqq3ge4ixwle + tests/test_superforcaster_full_search.py: bafybeieu6bylxkm7gsjagzbofn4vhx2a4ueyldwkboiso3g26en5ydk7uq fingerprint_ignore_patterns: [] entry_point: superforcaster_full_search.py callable: run diff --git a/packages/valory/customs/superforcaster_full_search/superforcaster_full_search.py b/packages/valory/customs/superforcaster_full_search/superforcaster_full_search.py index bf406833a..f3883e9ca 100644 --- a/packages/valory/customs/superforcaster_full_search/superforcaster_full_search.py +++ b/packages/valory/customs/superforcaster_full_search/superforcaster_full_search.py @@ -24,7 +24,17 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import openai import requests @@ -42,6 +52,9 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 def with_key_rotation(func: Callable) -> Callable: @@ -553,16 +566,249 @@ def _cap_evidence_block( return rendered -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: # noqa: BLE001 - print(f"Error extracting question: {e}") - question = prompt - return question +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc + + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[superforcaster_full_search] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + # Off-chain markers distinguishing a flagged null from a genuine + # max-uncertainty forecast and recording the derivation tier + # (matches superforcaster-polymarket-v4). + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params @with_key_rotation @@ -610,37 +856,104 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: today = date.today() d = today.strftime("%d/%m/%Y") - question = extract_question(prompt) + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[superforcaster_full_search] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[superforcaster_full_search] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[superforcaster_full_search] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) if source_content is not None: print("Using provided source content (cached replay)...") captured_source_content = source_content serper_data = source_content.get("serper_response", source_content) + organic_data, misc_data = _shape_serper_sources( + serper_data, "cached replay" + ) # Shallow-copy each organic item so attaching `content` does not # mutate the caller's cached source_content payload. - organic_data = [ - dict(it) for it in serper_data.get("organic", [])[:MAX_SOURCES] - ] - misc_data = serper_data.get("peopleAlsoAsk", []) + organic_data = [dict(it) for it in organic_data] + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="cached replay", + tier=tier, + scan_truncated=scan_truncated, + ) cached_pages = source_content.get("pages", {}) cached_mode = source_content.get("mode", source_content_mode) _hydrate_organic_from_pages(organic_data, cached_pages, cached_mode) sources = _cap_evidence_block(organic_data, misc_data, model) else: + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) serper_api_key = kwargs["api_keys"]["serperapi"] print("Fetching additional sources...") - serper_response = fetch_additional_sources(question, serper_api_key) + # Use the compressed search_query instead of the full prompt so + # Serper returns organic results for free-text callers (issue #455). + serper_response = fetch_additional_sources(search_query, serper_api_key) # Surface HTTP errors with a real status code instead of crashing # .json() on a non-JSON 4xx/5xx body (matches the fleet pattern). serper_response.raise_for_status() sources_data = serper_response.json() print(f"Additional sources fetched: {sources_data}") + organic_data, misc_data = _shape_serper_sources(sources_data, "live search") # Shallow-copy organic items: _scrape_pages attaches `content`, # and we don't want that leaking into the captured serper_response. - organic_data = [ - dict(it) for it in sources_data.get("organic", [])[:MAX_SOURCES] - ] - misc_data = sources_data.get("peopleAlsoAsk", []) + organic_data = [dict(it) for it in organic_data] + # Empty-retrieval guard: even a correct short query can fail + # (e.g. very niche or recent market). Placed BEFORE the scraping + # step so empty retrieval wastes no page fetches (issue #455). + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content={ + "mode": source_content_mode, + "serper_response": sources_data, + }, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) print("Scraping page content for top organic results...") captured_pages = _scrape_pages(organic_data, source_content_mode) print( @@ -679,6 +992,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = captured_source_content diff --git a/packages/valory/customs/superforcaster_full_search/tests/test_superforcaster_full_search.py b/packages/valory/customs/superforcaster_full_search/tests/test_superforcaster_full_search.py index 01a13e93f..f5dda529a 100644 --- a/packages/valory/customs/superforcaster_full_search/tests/test_superforcaster_full_search.py +++ b/packages/valory/customs/superforcaster_full_search/tests/test_superforcaster_full_search.py @@ -35,6 +35,7 @@ Usage, fetch_additional_sources, generate_prediction_with_retry, + parse_prompt, run, ) @@ -645,3 +646,211 @@ def test_max_cost_returns_float_not_wrapped_tuple(self) -> None: delivery_rate=0, ) assert result == 0.0123 + + +# Free-text format prompt with instruction boilerplate: the advertised input +# contract (issue #455) that the old extract_question collapsed into a +# prompt-shaped Serper query. +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently " + "transfer to Liverpool FC before the end of the summer 2025 transfer " + "window (September 2, 2025 23:59 UTC)? Resolution source: official club " + "announcements or BBC Sport. The market resolves YES if a permanent " + "transfer (not a loan) is confirmed by the resolution source before the " + "deadline." +) + +EMPTY_SERPER_RESPONSE: dict = {"organic": [], "peopleAlsoAsk": []} + + +class TestIssue455ParsePrompt: + """parse_prompt() -> (question_for_llm, search_query, tier).""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template parity: the bare question serves as both values.""" + question, query, tier = parse_prompt(PREDICTION_PROMPT) + assert tier == "template" + assert question == "Will X happen?" + assert query == question + + def test_free_text_llm_gets_full_prompt(self) -> None: + """Free-text input: the LLM question is the whole prompt.""" + question, _, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert tier == "clause" + assert question == LONG_FREE_TEXT_PROMPT + + def test_boilerplate_prefix_is_dropped_from_query(self) -> None: + """The query anchors at the market question, dropping instruction text.""" + _, query, _ = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert len(query) <= module._MAX_SEARCH_QUERY_LEN + + +class TestIssue455EmptyRetrievalGuard: + """Degenerate prompts and zero-hit retrieval return the flagged null.""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_degenerate_prompt_short_circuits_before_serper( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock, degenerate: str + ) -> None: + """Prompts with no searchable content never reach Serper at all.""" + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=degenerate, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + mock_fetch.assert_not_called() + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "empty query" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}._fetch_page_content") + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_zero_hit_live_returns_flagged_null_before_scrape( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + mock_page_fetch: MagicMock, + ) -> None: + """Both-empty live retrieval -> flagged null; no page is scraped.""" + mock_response = MagicMock() + mock_response.json.return_value = EMPTY_SERPER_RESPONSE + mock_fetch.return_value = mock_response + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + mock_page_fetch.assert_not_called() + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "live search" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + def test_empty_cached_replay_returns_flagged_null( + self, mock_client_mgr: MagicMock + ) -> None: + """Both-empty cached retrieval -> flagged null with the replay reason.""" + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + source_content={"serper_response": EMPTY_SERPER_RESPONSE}, + ) + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "cached replay" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_malformed_serper_body_is_error_not_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """organic: null is a broken integration -> typed error, not a null forecast.""" + mock_response = MagicMock() + mock_response.json.return_value = {"organic": None, "peopleAlsoAsk": []} + mock_fetch.return_value = mock_response + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + payload = json.loads(result[0]) + assert payload["p_yes"] is None + assert "organic" in payload["error"] + + +class TestIssue455RunWiring: + """run() feeds parsed.query to Serper and parsed.question to the LLM.""" + + @patch(f"{SF_MODULE}._fetch_page_content", side_effect=_fake_fetch) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_trader_request_sends_extracted_question_to_serper( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + _mock_page_fetch: MagicMock, + ) -> None: + """Template path parity: Serper gets the bare question (old behavior).""" + mock_response = MagicMock() + mock_response.json.return_value = FAKE_SERPER_RESPONSE + mock_fetch.return_value = mock_response + _stub_openai(mock_client_mgr) + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + mock_fetch.assert_called_once_with("Will X happen?", "serper-test") + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}._fetch_page_content", side_effect=_fake_fetch) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_free_text_serper_gets_query_llm_gets_full_prompt( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + _mock_page_fetch: MagicMock, + ) -> None: + """Free-text path: short query to Serper, whole prompt to the LLM.""" + mock_response = MagicMock() + mock_response.json.return_value = FAKE_SERPER_RESPONSE + mock_fetch.return_value = mock_response + _stub_openai(mock_client_mgr) + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + sent_query = mock_fetch.call_args[0][0] + assert sent_query.startswith("Will Alexander Isak") + assert len(sent_query) <= module._MAX_SEARCH_QUERY_LEN + # criteria text that the derived query drops must still reach the LLM + assert "official club announcements or BBC Sport" in result[1] + assert result[4]["parse_tier"] == "clause" + + @patch(f"{SF_MODULE}._fetch_page_content", side_effect=_fake_fetch) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_long_template_prompt_is_not_marked_truncated( + self, + mock_fetch: MagicMock, + mock_client_mgr: MagicMock, + _mock_page_fetch: MagicMock, + ) -> None: + """Template past the window is NOT flagged: question precedes the scan.""" + mock_response = MagicMock() + mock_response.json.return_value = FAKE_SERPER_RESPONSE + mock_fetch.return_value = mock_response + _stub_openai(mock_client_mgr) + prompt = PREDICTION_PROMPT + " filler" * (module._MAX_SCAN_CHARS // 3) + assert len(prompt) > module._MAX_SCAN_CHARS + result = run( + tool="superforcaster_full_search", + model="gpt-4o", + prompt=prompt, + api_keys=_make_mock_api_keys("false"), + counter_callback=None, + ) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False diff --git a/packages/valory/customs/superforcaster_polymarket_v1/component.yaml b/packages/valory/customs/superforcaster_polymarket_v1/component.yaml index 2bd097c2a..031e78dcd 100644 --- a/packages/valory/customs/superforcaster_polymarket_v1/component.yaml +++ b/packages/valory/customs/superforcaster_polymarket_v1/component.yaml @@ -8,9 +8,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeidr6ok7hgnywgtlggdajgklohlpaxnxmvpo2wn5fayq7fhzrr2sli - superforcaster_polymarket_v1.py: bafybeidaidqvhcx7ltfuolniqityheydug45apzpy66bhwfymn2qpl245i + superforcaster_polymarket_v1.py: bafybeihvxmlgtt67ucgnnexquflb6bv77p2vlx4v2ojlobzbaxst6rjwga tests/__init__.py: bafybeidegv7yfxpdytmvepvcqquzawanqjczrqdp2cesxxdx5vvdfmdgue - tests/test_superforcaster.py: bafybeiecpixajbp4vxbt2uuu2etzrjguvcbm4jh2m5h4mhx4pgjb46qtg4 + tests/test_superforcaster.py: bafybeie52evh6a6ykn2zzq45kgidhzpfrt7gnimkziq6uvusn5rxs2fovm fingerprint_ignore_patterns: [] entry_point: superforcaster_polymarket_v1.py callable: run diff --git a/packages/valory/customs/superforcaster_polymarket_v1/superforcaster_polymarket_v1.py b/packages/valory/customs/superforcaster_polymarket_v1/superforcaster_polymarket_v1.py index 8f3209016..77f90f953 100644 --- a/packages/valory/customs/superforcaster_polymarket_v1/superforcaster_polymarket_v1.py +++ b/packages/valory/customs/superforcaster_polymarket_v1/superforcaster_polymarket_v1.py @@ -23,7 +23,17 @@ import re import time from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import openai import requests @@ -39,6 +49,65 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). The on-chain JSON carries + only the four standard fields; the explicit marker for requesters lives in + used_params["empty_retrieval"] (off-chain metadata.params). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[superforcaster-polymarket-v1] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params def with_key_rotation(func: Callable) -> Callable: @@ -347,16 +416,192 @@ def format_sources_data(organic_data: Any, misc_data: Any) -> str: return sources -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: - print(f"Error extracting question: {e}") - question = prompt - return question +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc + + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) @with_key_rotation @@ -404,19 +649,73 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: today = date.today() d = today.strftime("%d/%m/%Y") - question = extract_question(prompt) + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[superforcaster-polymarket-v1] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[superforcaster-polymarket-v1] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[superforcaster-polymarket-v1] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) if source_content is not None: print("Using provided source content (cached replay)...") captured_source_content = source_content serper_data = source_content.get("serper_response", source_content) - organic_data = serper_data.get("organic", [])[:MAX_SOURCES] - misc_data = serper_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources( + serper_data, "cached replay" + ) + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="cached replay", + tier=tier, + scan_truncated=scan_truncated, + ) sources = format_sources_data(organic_data, misc_data) else: + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) serper_api_key = kwargs["api_keys"]["serperapi"] print("Fetching additional sources...") - serper_response = fetch_additional_sources(question, serper_api_key) + serper_response = fetch_additional_sources(search_query, serper_api_key) + # Raise on a 4xx/5xx error body (credit / auth error) instead of + # calling .json() on it and feeding the model an empty + # block that looks like a healthy run (matches the fleet pattern). + serper_response.raise_for_status() sources_data = serper_response.json() # mode tag included for consistency across tools; content is identical # regardless of mode since Serper returns structured JSON, not HTML @@ -425,8 +724,19 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "serper_response": sources_data, } print(f"Additional sources fetched: {sources_data}") - organic_data = sources_data.get("organic", [])[:MAX_SOURCES] - misc_data = sources_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources(sources_data, "live search") + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) print("Formating sources...") sources = format_sources_data(organic_data, misc_data) @@ -455,6 +765,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = captured_source_content diff --git a/packages/valory/customs/superforcaster_polymarket_v1/tests/test_superforcaster.py b/packages/valory/customs/superforcaster_polymarket_v1/tests/test_superforcaster.py index 731001d46..12e6deb85 100644 --- a/packages/valory/customs/superforcaster_polymarket_v1/tests/test_superforcaster.py +++ b/packages/valory/customs/superforcaster_polymarket_v1/tests/test_superforcaster.py @@ -30,6 +30,7 @@ from packages.valory.customs.superforcaster_polymarket_v1.superforcaster_polymarket_v1 import ( OpenAIClientManager, generate_prediction_with_retry, + parse_prompt, run, ) @@ -208,3 +209,199 @@ def test_flag_off_no_source_content( assert result[0] == PREDICTION_JSON used_params = result[4] assert "source_content" not in used_params + + +EMPTY_SERPER_RESPONSE: dict = {"organic": [], "peopleAlsoAsk": []} + +# Free-text format prompt: the advertised contract (issue #455) +FREE_TEXT_PROMPT = "Will Alexander Isak join Liverpool before September 2 2025?" +# Long free-text prompt that would return empty Serper results if passed raw +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently transfer " + "to Liverpool FC before the end of the summer 2025 transfer window (September 2, " + "2025 23:59 UTC)? Resolution source: official club announcements or BBC Sport. " + "The market resolves YES if a permanent transfer (not a loan) is confirmed by " + "the resolution source before the deadline." +) + + +class TestParsePromptContract: + """parse_prompt() -> (question_for_llm, search_query, tier).""" + + def test_trader_template_parity(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = parse_prompt(PREDICTION_PROMPT) + assert tier == "template" + assert question == "Will X happen?" + assert query == question + + def test_free_text_clause_derivation(self) -> None: + """Boilerplate lead-in anchors the market question; LLM sees everything.""" + question, query, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert tier == "clause" + assert question == LONG_FREE_TEXT_PROMPT + # "Please predict the following market: " is gone; the clause survives + # whole, including the deadline and the trailing '?'. + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert len(query) <= module._MAX_SEARCH_QUERY_LEN + + +class TestIssue455Guards: + """Empty-query short-circuit and both-empty-retrieval flagged nulls.""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_degenerate_prompt_short_circuits_before_serper( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock, degenerate: str + ) -> None: + """Prompts with no searchable content never reach Serper at all.""" + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=degenerate, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_fetch.assert_not_called() + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["p_no"] == 0.5 + assert parsed["confidence"] == 0.0 and parsed["info_utility"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "empty query" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_both_empty_live_retrieval_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """A genuine zero-hit returns the flagged null, LLM never called.""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["confidence"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "live search" + assert result[4]["parse_tier"] == "clause" + mock_client.completions.assert_not_called() + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_cached_replay_both_empty_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """An empty cached source_content returns the flagged null too.""" + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + source_content={"serper_response": EMPTY_SERPER_RESPONSE}, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["confidence"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "cached replay" + mock_fetch.assert_not_called() + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_reshaped_serper_body_is_an_error_not_a_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """A 200 body without the organic key surfaces as an error, not 0.5.""" + mock_fetch.return_value = MagicMock(json=lambda: {"message": "quota exceeded"}) + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + # v1's failure contract returns the exception string, never 0.5/0.5. + assert result[0].startswith("live search:") + assert "organic" in result[0] + + +class TestIssue455RunWiring: + """run() feeds the LLM the parsed question and Serper the derived query.""" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_trader_template_llm_and_serper_parity( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Trader path parity: the bare extracted question feeds both sinks.""" + serper_resp = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_fetch.return_value = serper_resp + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + assert mock_fetch.call_args[0][0] == "Will X happen?" + # the HTTP-error guard must actually run on the happy path + serper_resp.raise_for_status.assert_called_once() + # the LLM prompt carries the bare question, not the full template + assert "Will X happen?" in result[1] + assert "`yes` option" not in result[1] + assert result[4]["parse_tier"] == "template" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_free_text_serper_gets_short_query_llm_gets_full_prompt( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Free text: Serper gets the derived query, the LLM the whole prompt.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + query_sent = mock_fetch.call_args[0][0] + assert len(query_sent) <= module._MAX_SEARCH_QUERY_LEN + assert query_sent != LONG_FREE_TEXT_PROMPT + # criteria text the derived query drops must still reach the LLM + assert "official club announcements or BBC Sport" in result[1] + assert result[4]["parse_tier"] == "clause" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_long_template_prompt_is_not_marked_truncated( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Template past the window is NOT flagged: it precedes the scan.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + _install_mock_client(mock_client_mgr) + prompt = PREDICTION_PROMPT + " filler" * (module._MAX_SCAN_CHARS // 3) + assert len(prompt) > module._MAX_SCAN_CHARS + result = run( + tool="superforcaster-polymarket-v1", + model="gpt-4o", + prompt=prompt, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False diff --git a/packages/valory/customs/superforcaster_polymarket_v2/component.yaml b/packages/valory/customs/superforcaster_polymarket_v2/component.yaml index 67ba20260..c89ed6f08 100644 --- a/packages/valory/customs/superforcaster_polymarket_v2/component.yaml +++ b/packages/valory/customs/superforcaster_polymarket_v2/component.yaml @@ -9,9 +9,9 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeidr6ok7hgnywgtlggdajgklohlpaxnxmvpo2wn5fayq7fhzrr2sli - superforcaster_polymarket_v2.py: bafybeiergairblrcn6idlq6jwtids3ooq5wulsosejcqnu7tzoraypgtvu + superforcaster_polymarket_v2.py: bafybeihqyc67hp7vt4rcgrjuxnb46bsglkwik267ou7gjzqlzygfcqk6am tests/__init__.py: bafybeidegv7yfxpdytmvepvcqquzawanqjczrqdp2cesxxdx5vvdfmdgue - tests/test_superforcaster.py: bafybeibms6bcbtyhat4fxem6y6lglkzimqgijhxfdfdjadkj45jnyekj44 + tests/test_superforcaster.py: bafybeibg2o5a6a5yooxtyophjbkkqyhutw4hwr5ubyi2qheviyaqtl3pja fingerprint_ignore_patterns: [] entry_point: superforcaster_polymarket_v2.py callable: run diff --git a/packages/valory/customs/superforcaster_polymarket_v2/superforcaster_polymarket_v2.py b/packages/valory/customs/superforcaster_polymarket_v2/superforcaster_polymarket_v2.py index 2cbe03094..b8d1921f1 100644 --- a/packages/valory/customs/superforcaster_polymarket_v2/superforcaster_polymarket_v2.py +++ b/packages/valory/customs/superforcaster_polymarket_v2/superforcaster_polymarket_v2.py @@ -23,7 +23,17 @@ import re import time from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import openai import requests @@ -39,6 +49,9 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 def with_key_rotation(func: Callable) -> Callable: @@ -355,16 +368,248 @@ def format_sources_data(organic_data: Any, misc_data: Any) -> str: return sources -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: - print(f"Error extracting question: {e}") - question = prompt - return question +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc + + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). The on-chain JSON carries + only the four standard fields; the explicit marker for requesters lives + in used_params["empty_retrieval"] (off-chain metadata.params). + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[superforcaster-polymarket-v2] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params @with_key_rotation @@ -412,19 +657,73 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: today = date.today() d = today.strftime("%d/%m/%Y") - question = extract_question(prompt) + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[superforcaster-polymarket-v2] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[superforcaster-polymarket-v2] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[superforcaster-polymarket-v2] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) if source_content is not None: print("Using provided source content (cached replay)...") captured_source_content = source_content serper_data = source_content.get("serper_response", source_content) - organic_data = serper_data.get("organic", [])[:MAX_SOURCES] - misc_data = serper_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources( + serper_data, "cached replay" + ) + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="cached replay", + tier=tier, + scan_truncated=scan_truncated, + ) sources = format_sources_data(organic_data, misc_data) else: + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) serper_api_key = kwargs["api_keys"]["serperapi"] print("Fetching additional sources...") - serper_response = fetch_additional_sources(question, serper_api_key) + serper_response = fetch_additional_sources(search_query, serper_api_key) + # Raise on a 4xx/5xx error body (credit / auth error) instead of + # calling .json() on it and feeding the model an empty + # block that looks like a healthy run (matches the fleet pattern). + serper_response.raise_for_status() sources_data = serper_response.json() # mode tag included for consistency across tools; content is identical # regardless of mode since Serper returns structured JSON, not HTML @@ -433,8 +732,19 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "serper_response": sources_data, } print(f"Additional sources fetched: {sources_data}") - organic_data = sources_data.get("organic", [])[:MAX_SOURCES] - misc_data = sources_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources(sources_data, "live search") + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) print("Formating sources...") sources = format_sources_data(organic_data, misc_data) @@ -463,6 +773,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = captured_source_content diff --git a/packages/valory/customs/superforcaster_polymarket_v2/tests/test_superforcaster.py b/packages/valory/customs/superforcaster_polymarket_v2/tests/test_superforcaster.py index 8dc907708..688db81cd 100644 --- a/packages/valory/customs/superforcaster_polymarket_v2/tests/test_superforcaster.py +++ b/packages/valory/customs/superforcaster_polymarket_v2/tests/test_superforcaster.py @@ -30,6 +30,7 @@ from packages.valory.customs.superforcaster_polymarket_v2.superforcaster_polymarket_v2 import ( OpenAIClientManager, generate_prediction_with_retry, + parse_prompt, run, ) @@ -208,3 +209,219 @@ def test_flag_off_no_source_content( assert result[0] == PREDICTION_JSON used_params = result[4] assert "source_content" not in used_params + + +EMPTY_SERPER_RESPONSE: dict = {"organic": [], "peopleAlsoAsk": []} + +# Free-text format prompts: the advertised contract (issue #455) +FREE_TEXT_PROMPT = "Will Alexander Isak join Liverpool before September 2 2025?" +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently transfer " + "to Liverpool FC before the end of the summer 2025 transfer window (September 2, " + "2025 23:59 UTC)? Resolution source: official club announcements or BBC Sport. " + "The market resolves YES if a permanent transfer (not a loan) is confirmed by " + "the resolution source before the deadline." +) + + +class TestParsePromptPort: + """parse_prompt() -> (question_for_llm, search_query, tier) (issue #455 port).""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = parse_prompt(PREDICTION_PROMPT) + assert question == "Will X happen?" + assert query == question + assert tier == "template" + + def test_free_text_llm_gets_full_prompt(self) -> None: + """Free-text input: the LLM question is the whole prompt.""" + question, _, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert question == LONG_FREE_TEXT_PROMPT + assert tier == "clause" + + def test_boilerplate_prefix_is_dropped_from_query(self) -> None: + """The query anchors at the market question, dropping instruction text.""" + _, query, _ = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert len(query) <= module._MAX_SEARCH_QUERY_LEN + + def test_double_quotes_are_stripped_from_query_only(self) -> None: + """Quoted spans become exact-match Serper terms; drop them from the query.""" + prompt = ( + 'Will any candle have a final "High" price >= 82000 in the window? ' + "Resolution source: Binance." + ) + question, query, _ = parse_prompt(prompt) + assert '"' not in query + assert "High" in query + assert '"High"' in question # the LLM still sees the exact wording + + def test_no_question_clause_truncates(self) -> None: + """A prompt with no question clause falls back to the capped prompt.""" + no_q = "x" * 300 + question, query, tier = parse_prompt(no_q) + assert question == no_q + assert tier == "raw" + assert len(query) == module._MAX_SEARCH_QUERY_LEN + + +class TestIssue455Guards: + """Short-circuit, empty-retrieval flagged nulls, and parity (issue #455).""" + + @pytest.mark.parametrize("degenerate", ["", " ", "???", '"""']) + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_degenerate_prompt_short_circuits_before_search( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock, degenerate: str + ) -> None: + """Prompts with no searchable content never reach Serper or the LLM.""" + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=degenerate, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_fetch.assert_not_called() + mock_client.completions.assert_not_called() + assert json.loads(result[0]) == { + "p_yes": 0.5, + "p_no": 0.5, + "confidence": 0.0, + "info_utility": 0.0, + } + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "empty query" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_zero_hit_live_search_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Organic AND peopleAlsoAsk both empty -> flagged null, reason 'live search'.""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_fetch.assert_called_once() + mock_client.completions.assert_not_called() + assert json.loads(result[0])["p_yes"] == 0.5 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "live search" + assert result[4]["parse_tier"] == "clause" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + def test_empty_cached_replay_returns_flagged_null( + self, mock_client_mgr: MagicMock + ) -> None: + """An empty cached capture replays to the same flagged null.""" + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + source_content={"serper_response": EMPTY_SERPER_RESPONSE}, + ) + assert json.loads(result[0])["confidence"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "cached replay" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_trader_template_parity_end_to_end( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """LLM-input parity: the template path feeds the extracted question to both sinks.""" + mock_serper = MagicMock() + mock_serper.json.return_value = FAKE_SERPER_RESPONSE + mock_fetch.return_value = mock_serper + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=PREDICTION_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_fetch.assert_called_once_with("Will X happen?", "serper-test") + assert "Question:\nWill X happen?" in result[1] + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_long_template_prompt_is_not_marked_truncated( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Template past the scan window is NOT flagged: the match precedes the scan.""" + mock_serper = MagicMock() + mock_serper.json.return_value = FAKE_SERPER_RESPONSE + mock_fetch.return_value = mock_serper + _install_mock_client(mock_client_mgr) + prompt = PREDICTION_PROMPT + " filler" * (module._MAX_SCAN_CHARS // 3) + assert len(prompt) > module._MAX_SCAN_CHARS + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=prompt, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_free_text_search_uses_derived_query_llm_gets_full_prompt( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Serper gets the short derived query; the LLM still sees the whole prompt.""" + mock_serper = MagicMock() + mock_serper.json.return_value = FAKE_SERPER_RESPONSE + mock_fetch.return_value = mock_serper + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + sent_query = mock_fetch.call_args[0][0] + assert sent_query.startswith("Will Alexander Isak") + assert len(sent_query) <= module._MAX_SEARCH_QUERY_LEN + # criteria text the derived query drops must still reach the LLM + assert "official club announcements or BBC Sport" in result[1] + assert result[4]["parse_tier"] == "clause" + + @patch(f"{SF_MODULE}.OpenAIClientManager") + @patch(f"{SF_MODULE}.fetch_additional_sources") + def test_malformed_serper_body_raises_typed_error( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """organic: null is a broken integration -> typed error, not a flagged null.""" + mock_fetch.return_value = MagicMock( + json=lambda: {"organic": None, "peopleAlsoAsk": []} + ) + _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v2", + model="gpt-4o", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + # v2's key-rotation wrapper surfaces the error string as result[0] + assert "'organic'" in result[0] + assert result[4] is None diff --git a/packages/valory/customs/superforcaster_polymarket_v3/component.yaml b/packages/valory/customs/superforcaster_polymarket_v3/component.yaml index 7f8a17365..8e69ba3d8 100644 --- a/packages/valory/customs/superforcaster_polymarket_v3/component.yaml +++ b/packages/valory/customs/superforcaster_polymarket_v3/component.yaml @@ -15,9 +15,10 @@ license: Apache-2.0 aea_version: '>=1.0.0, <2.0.0' fingerprint: __init__.py: bafybeidr6ok7hgnywgtlggdajgklohlpaxnxmvpo2wn5fayq7fhzrr2sli - superforcaster_polymarket_v3.py: bafybeidcea6xdiakikjwm5dutimyxhonldnctbwiq2dgekhpyexhmrw47m + superforcaster_polymarket_v3.py: bafybeibvkddmrcqgdxk6txwy3mx3erjuqzlv44eqrgsyrjrckets2sqjt4 tests/__init__.py: bafybeidegv7yfxpdytmvepvcqquzawanqjczrqdp2cesxxdx5vvdfmdgue - tests/test_superforcaster.py: bafybeidhzamokjfkii4v5xgfk54j3huvn3lgvcmqny7rs74otggahxhg54 + tests/test_superforcaster.py: bafybeidhaql6gs65svanpc5dh2enogfi4e2jubtlotst4xwxbromq3iu6q + tests/test_superforcaster_polymarket_v3.py: bafybeige3b2ywoo562rt74a4r4tpxge3lzbjabf3fxnygpp525x2ytt6di fingerprint_ignore_patterns: [] entry_point: superforcaster_polymarket_v3.py callable: run diff --git a/packages/valory/customs/superforcaster_polymarket_v3/superforcaster_polymarket_v3.py b/packages/valory/customs/superforcaster_polymarket_v3/superforcaster_polymarket_v3.py index f7e98dca3..ffcf81ce5 100644 --- a/packages/valory/customs/superforcaster_polymarket_v3/superforcaster_polymarket_v3.py +++ b/packages/valory/customs/superforcaster_polymarket_v3/superforcaster_polymarket_v3.py @@ -59,7 +59,17 @@ import re import time from datetime import date -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + NamedTuple, + Optional, + Tuple, + Union, +) import anthropic import openai @@ -77,6 +87,9 @@ N_MODEL_CALLS = 1 DEFAULT_DELIVERY_RATE = 100 +# Serper degrades sharply on prompt-shaped queries (instruction boilerplate, +# JSON-format text), in the worst case to zero organic results (issue #455). +_MAX_SEARCH_QUERY_LEN = 150 # Anthropic exception classes the rotation branch treats as recoverable. @@ -594,16 +607,249 @@ def format_sources_data(organic_data: Any, misc_data: Any) -> str: return sources -def extract_question(prompt: str) -> str: - """Uses regexp to extract question from the prompt""" - # Match from 'question "' to '" and the `yes`' to handle nested quotes - pattern = r'question\s+"(.+?)"\s+and\s+the\s+`yes`' - try: - question = re.findall(pattern, prompt, re.DOTALL)[0] - except Exception as e: - print(f"Error extracting question: {e}") - question = prompt - return question +# Matches from 'question "' to '" and the `yes`' to handle nested quotes. +_TRADER_TEMPLATE_RE = re.compile(r'question\s+"(.+?)"\s+and\s+the\s+`yes`', re.DOTALL) +# Question-clause candidates: every question-word occurrence starts one, running +# to the FIRST '?' after it (via str.find; tolerates embedded dots -- +# abbreviations, decimals, market ids -- which sentence-boundary splitting +# would cut on). +# Candidates may overlap; a feature score selects the market question among +# them (see _score_clause). +_QUESTION_WORD_RE = re.compile( + r"(?:will|is|are|was|were|does|do|did|can|could|who|what|when|where|which" + r"|how|whether)\b", + re.IGNORECASE, +) +# Meta/instruction stems: a question addressed at the RESPONDER ("Can you +# estimate...", "What is your probability...") or prompt scaffolding ("What +# follows is..."), never the market question itself. Second-person only: +# first-person clauses ("Will we...", "Do I...") occur in real market wording. +_META_STEM_RE = re.compile( + r"^(?:(?:can|could|would|will|do|does|did|is|are)\s+(?:you|your)\b" + r"|what\s+(?:is|are)\s+(?:your|the\s+(?:respective\s+)?probabilit)" + r"|what\s+follows\b)", + re.IGNORECASE, +) +# Deliberately case-sensitive (unlike the IGNORECASE _QUESTION_WORD_RE): a +# capitalized market verb marks a sentence-initial market question, and adding +# IGNORECASE here would double-count lowercase occurrences via the +1 bonus. +_MARKET_VERB_RE = re.compile( + r"^(?:Will|Is|Are|Was|Were|Does|Do|Did|Which|Who|When|Whether)\b" +) +# Chars that may directly precede a sentence-initial question word: whitespace, +# sentence punctuation, ASCII quotes/paren, and typographic quotes. +_CLAUSE_BOUNDARY = " \t\n.!?:\"'(\u201c\u201d\u2018\u2019" +# Candidate scanning is bounded to the prompt head: every question-word +# occurrence starts a candidate and each candidate scans forward for '?', so +# an unbounded scan is quadratic. Measured cost is small at the mech's cap +# (~6.6ms unbounded at 100KB, the MAX_PROMPT_BYTES limit in the mech repo's +# valory/task_execution skill) but grows ~4x per 2x and benchmark/direct +# calls are not capped at all (multi-MB prompts reach seconds) -- the window +# is defence-in-depth for those paths. Market questions sit in the prompt +# head in practice (the longest observed production prompt is under 1KB), so +# a 10KB window loses nothing on real traffic. +_MAX_SCAN_CHARS = 10_000 +# Near-best window for the last-market-verb tiebreaker. Equals the largest +# single-feature weight (the digit bonus in _score_clause) so a market clause +# can never be pushed out of contention by one feature alone. +_NEAR_BEST_WINDOW = 3 + + +def _score_clause(prompt: str, start: int, clause: str) -> int: + """Score a question-clause candidate; the market question should win. + + Features: digits (market questions carry deadlines/quantities; instruction + and clarifying questions rarely do), a market-shaped opening verb, a + sentence-initial capitalized start, a penalty for responder-addressed / + scaffolding stems, and a penalty for sweeping across a sentence boundary. + + :param prompt: the full prompt (for boundary context). + :param start: the clause's start offset in the prompt. + :param clause: the candidate clause text. + :return: the feature score (higher = more market-question-shaped). + """ + score = 0 + if any(ch.isdigit() for ch in clause): + score += 3 + if _MARKET_VERB_RE.match(clause): + score += 1 + if clause[0].isupper() and (start == 0 or prompt[start - 1] in _CLAUSE_BOUNDARY): + score += 2 + if _META_STEM_RE.match(clause): + score -= 3 + if ". " in clause: + score -= 1 + return score + + +def _shape_serper_sources( + raw: Dict[str, Any], context: str +) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + """Validate a serper_response body and slice it into (organic, misc). + + A body without the organic key is a broken or reshaped integration (a + quota-error body, a renamed key, a corrupted cache entry), not a genuine + zero-hit -- raise so it surfaces as an error null with error_type instead + of collapsing into the flagged null. + + :param raw: the serper_response dict (live or cached). + :param context: short label for the error message (live vs cached replay). + :return: the (organic, peopleAlsoAsk) lists, organic capped at MAX_SOURCES. + """ + if not isinstance(raw.get("organic"), list): + raise ValueError( + f"{context}: Serper response missing or malformed 'organic' key; " + f"got keys: {sorted(raw)[:8]}" + ) + misc = raw.get("peopleAlsoAsk", []) + if not isinstance(misc, list): + raise ValueError( + f"{context}: Serper response has a malformed 'peopleAlsoAsk' key; " + f"got {type(misc).__name__}" + ) + return raw["organic"][:MAX_SOURCES], misc + + +def _truncate_query(query: str) -> str: + """Cap the query at _MAX_SEARCH_QUERY_LEN, cutting on a word boundary. + + :param query: the derived search query. + :return: the query, truncated without a dangling partial word. + """ + if len(query) <= _MAX_SEARCH_QUERY_LEN: + return query + cut = query[:_MAX_SEARCH_QUERY_LEN] + if not query[_MAX_SEARCH_QUERY_LEN].isspace() and not cut.endswith(" "): + cut = cut.rsplit(None, 1)[0] if " " in cut else cut + return cut.rstrip() + + +class ParsedPrompt(NamedTuple): + """parse_prompt's result: the LLM question, the Serper query, the tier.""" + + question: str + query: str + tier: Literal["template", "clause", "raw"] + + +def parse_prompt(prompt: str) -> ParsedPrompt: + """Split a request prompt into the LLM question and the Serper search query. + + Trader-template prompts carry the bare market question between known + delimiters: it serves as both values, keeping that path byte-identical to + previous releases. Any other prompt is free text under the advertised + input contract (issue #455): the LLM receives the WHOLE prompt (resolution + criteria, source, and deadline stay in context) while the search query is + the best-scoring question clause (see _score_clause), with double quotes + dropped (Serper treats quoted spans as exact-match terms) and the length + capped on a word boundary. + + :param prompt: the raw prompt passed to run(). + :return: a ParsedPrompt -- tier is 'template' (trader regex matched), + 'clause' (a scored question clause), or 'raw' (no clause found; + capped prompt head). + """ + match = _TRADER_TEMPLATE_RE.findall(prompt) + if match: + question = match[0] + return ParsedPrompt(question, question, "template") + scan = prompt[:_MAX_SCAN_CHARS] + candidates = [] + for word in _QUESTION_WORD_RE.finditer(scan): + start = word.start() + if start > 0 and scan[start - 1].isalnum(): + continue + end = scan.find("?", start) + if end == -1: + continue + clause = scan[start : end + 1] + candidates.append( + (_score_clause(scan, start, clause), len(clause), -start, clause) + ) + tier: Literal["template", "clause", "raw"] + if candidates: + # Clarifying questions (inside resolution criteria) often carry the + # dates/counts that outscore a digit-free market question. In free + # text the market question is reliably the LAST market-verb-shaped + # question -- clarifiers and instructions precede it -- so among + # candidates near the best score, prefer the last market-verb one. + best_score = max(candidates)[0] + market_shaped = [ + c + for c in candidates + if c[0] >= best_score - _NEAR_BEST_WINDOW + and _MARKET_VERB_RE.match(c[3]) + and not _META_STEM_RE.match(c[3]) + ] + chosen = ( + min(market_shaped, key=lambda c: c[2]) if market_shaped else max(candidates) + ) + query, tier = chosen[3], "clause" + else: + query, tier = scan, "raw" + query = _truncate_query(query.replace('"', "").strip()) + if not query: + # Degenerate prompts (only quotes/whitespace) must not strip down to + # an empty Serper query -- fall back to the unstripped prompt head. + query = _truncate_query(prompt.strip()) + return ParsedPrompt(prompt, query, tier) + + +def _flagged_null_result( + *, + model: str, + temperature: float, + max_tokens: int, + captured_source_content: Optional[Dict[str, Any]], + return_source_content: bool, + counter_callback: Optional[Callable[..., Any]], + context: str, + tier: str, + scan_truncated: bool = False, +) -> MechResponse: + """Build the flagged null prediction returned on empty retrieval. + + A VALID prediction (p_yes = p_no = 0.5) with zero confidence and + info_utility, so a requester can detect and discount it while the strict + trader consumer still parses it (issue #455). The on-chain JSON carries + only the four standard fields; the explicit marker for requesters lives + in used_params["empty_retrieval"] (off-chain metadata.params), matching + superforcaster-polymarket-v4. + + :param model: the model name recorded in used_params. + :param temperature: the temperature recorded in used_params. + :param max_tokens: the max_tokens recorded in used_params. + :param captured_source_content: the (empty) retrieval capture. + :param return_source_content: whether to attach the capture to used_params. + :param counter_callback: the cost callback, threaded back unchanged. + :param context: why the null was produced; recorded unconditionally in + used_params["null_reason"] so a skipped Serper call ("empty query") + stays distinguishable from a genuine zero-hit ("live search"). + :param tier: the parse_prompt tier that produced the search query. + :param scan_truncated: whether the scan window did not cover the whole + prompt (any non-template tier; a template match returns before the + window can matter). + :return: the flagged-null MechResponse tuple. + """ + print( + f"[superforcaster-polymarket-v3] {context}: empty retrieval" + " -- returning null prediction" + ) + null_result = json.dumps( + {"p_yes": 0.5, "p_no": 0.5, "confidence": 0.0, "info_utility": 0.0} + ) + used_params: Dict[str, Any] = { + "model": model, + "temperature": temperature, + "max_tokens": max_tokens, + "empty_retrieval": True, + "null_reason": context, + "parse_tier": tier, + "scan_truncated": scan_truncated, + } + if return_source_content: + used_params["source_content"] = captured_source_content + return null_result, "", None, counter_callback, used_params @with_key_rotation @@ -670,19 +916,73 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: today = date.today() d = today.strftime("%d/%m/%Y") - question = extract_question(prompt) + question, search_query, tier = parse_prompt(prompt) + # The scan window not covering the whole prompt is observable on its + # own: even a clause-tier pick may have missed the real question + # sitting past the window (not only the raw-tier no-clause case). + # A template match is exempt: it returns the exact question before + # the window plays any role, so nothing can have been missed. + scan_truncated = tier != "template" and len(prompt) > _MAX_SCAN_CHARS + if scan_truncated: + print( + f"[superforcaster-polymarket-v3] Scan window exhausted: " + f"prompt is {len(prompt)} chars, scanned the first " + f"{_MAX_SCAN_CHARS}; tier={tier}, query: {search_query!r}" + ) + elif tier == "raw": + print( + "[superforcaster-polymarket-v3] No question clause found; " + f"using capped prompt head as the search query: {search_query!r}" + ) + elif tier == "clause": + print( + f"[superforcaster-polymarket-v3] Free-text prompt (tier={tier}); " + f"derived search query: {search_query!r}" + ) if source_content is not None: print("Using provided source content (cached replay)...") captured_source_content = source_content serper_data = source_content.get("serper_response", source_content) - organic_data = serper_data.get("organic", [])[:MAX_SOURCES] - misc_data = serper_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources( + serper_data, "cached replay" + ) + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="cached replay", + tier=tier, + scan_truncated=scan_truncated, + ) sources = format_sources_data(organic_data, misc_data) else: + if not any(ch.isalnum() for ch in search_query): + # Nothing searchable: no alphanumeric character at all (empty, + # whitespace, quotes, or bare punctuation) -- skip the wasted + # Serper call and return the flagged null directly. + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=None, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="empty query", + tier=tier, + scan_truncated=scan_truncated, + ) serper_api_key = kwargs["api_keys"]["serperapi"] print("Fetching additional sources...") - serper_response = fetch_additional_sources(question, serper_api_key) + serper_response = fetch_additional_sources(search_query, serper_api_key) + # Raise on a 4xx/5xx error body (credit / auth error) instead of + # calling .json() on it and feeding the model an empty + # block that looks like a healthy run (matches the fleet pattern). + serper_response.raise_for_status() sources_data = serper_response.json() # mode tag included for consistency across tools; content is identical # regardless of mode since Serper returns structured JSON, not HTML @@ -691,8 +991,19 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "serper_response": sources_data, } print(f"Additional sources fetched: {sources_data}") - organic_data = sources_data.get("organic", [])[:MAX_SOURCES] - misc_data = sources_data.get("peopleAlsoAsk", []) + organic_data, misc_data = _shape_serper_sources(sources_data, "live search") + if not organic_data and not misc_data: + return _flagged_null_result( + model=model, + temperature=temperature, + max_tokens=max_tokens, + captured_source_content=captured_source_content, + return_source_content=return_source_content, + counter_callback=counter_callback, + context="live search", + tier=tier, + scan_truncated=scan_truncated, + ) print("Formating sources...") sources = format_sources_data(organic_data, misc_data) @@ -721,6 +1032,8 @@ def run(**kwargs: Any) -> Union[MaxCostResponse, MechResponse]: "model": model, "temperature": temperature, "max_tokens": max_tokens, + "parse_tier": tier, + "scan_truncated": scan_truncated, } if return_source_content: used_params["source_content"] = captured_source_content diff --git a/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster.py b/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster.py index f3b034bdd..695bb2d6c 100644 --- a/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster.py +++ b/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster.py @@ -666,8 +666,22 @@ def test_run_with_claude_fable_5_returns_valid_prediction(self) -> None: mock_llm_client.completions.return_value = mock_response MockManager.return_value.__enter__.return_value = mock_llm_client MockManager.return_value.__exit__.return_value = None + # Non-empty organic results: an all-empty retrieval now returns + # the flagged null prediction (issue #455) instead of calling + # the LLM, which is covered by its own tests. MockFetchSources.return_value = MagicMock( - json=MagicMock(return_value={"organic": []}) + json=MagicMock( + return_value={ + "organic": [ + { + "title": "T", + "link": "https://example.test", + "snippet": "S", + "position": 1, + } + ] + } + ) ) result = v3_module.run( diff --git a/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster_polymarket_v3.py b/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster_polymarket_v3.py new file mode 100644 index 000000000..5ad1440fe --- /dev/null +++ b/packages/valory/customs/superforcaster_polymarket_v3/tests/test_superforcaster_polymarket_v3.py @@ -0,0 +1,374 @@ +# -*- coding: utf-8 -*- +# ------------------------------------------------------------------------------ +# +# Copyright 2026 Valory AG +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# ------------------------------------------------------------------------------ + +"""Unit tests for superforcaster-polymarket-v3's free-text-input contract (issue #455).""" + +import json +from unittest.mock import MagicMock, patch + +import pytest + +from packages.valory.customs.superforcaster_polymarket_v3.superforcaster_polymarket_v3 import ( + _MAX_SCAN_CHARS, + _MAX_SEARCH_QUERY_LEN, + parse_prompt, + run, +) + +V3_MODULE = ( + "packages.valory.customs.superforcaster_polymarket_v3." + "superforcaster_polymarket_v3" +) + +FAKE_SERPER_RESPONSE = { + "organic": [{"title": "T", "link": "https://example.test", "snippet": "S"}], + "peopleAlsoAsk": [{"question": "Q?", "snippet": "A."}], +} + +EMPTY_SERPER_RESPONSE: dict = {"organic": [], "peopleAlsoAsk": []} + +PREDICTION_JSON = json.dumps( + {"p_yes": 0.6, "p_no": 0.4, "confidence": 0.8, "info_utility": 0.6} +) + +# Trader-template format prompt (regression: previous callers must still work) +TRADER_PROMPT = ( + 'Given the question "Will X happen?" and the `yes` answer criterion, ...' +) +# Free-text format prompt: the advertised contract (issue #455) +FREE_TEXT_PROMPT = "Will Alexander Isak join Liverpool before September 2 2025?" +# Long free-text prompt that would return empty Serper results if passed raw +LONG_FREE_TEXT_PROMPT = ( + "Please predict the following market: Will Alexander Isak permanently transfer " + "to Liverpool FC before the end of the summer 2025 transfer window (September 2, " + "2025 23:59 UTC)? Resolution source: official club announcements or BBC Sport. " + "The market resolves YES if a permanent transfer (not a loan) is confirmed by " + "the resolution source before the deadline." +) + + +def _make_mock_api_keys() -> MagicMock: + """Create a mock KeyChain-like api_keys object with both provider keys.""" + services = { + "openai": "sk-test", + "anthropic": "sk-ant-test", + "serperapi": "serper-test", + "return_source_content": "false", + "source_content_mode": "cleaned", + } + mock = MagicMock() + mock.__getitem__ = lambda self, key: services[key] + mock.get = lambda key, default="": services.get(key, default) + mock.max_retries = lambda: {"openai": 0, "openrouter": 0, "anthropic": 0} + return mock + + +def _install_mock_client(mock_client_mgr: MagicMock) -> MagicMock: + """Wire LLMClientManager to a client whose completions() yields PREDICTION_JSON. + + :param mock_client_mgr: the patched LLMClientManager mock. + :return: the inner mock client wired into the manager's __enter__. + """ + mock_client = MagicMock() + mock_response = MagicMock() + mock_response.content = PREDICTION_JSON + mock_response.usage.prompt_tokens = 10 + mock_response.usage.completion_tokens = 5 + mock_client.completions.return_value = mock_response + mock_client_mgr.return_value.__enter__ = MagicMock(return_value=mock_client) + mock_client_mgr.return_value.__exit__ = MagicMock(return_value=False) + return mock_client + + +class TestParsePrompt: + """parse_prompt() -> (question_for_llm, search_query, tier).""" + + def test_trader_template_uses_extracted_question_for_both(self) -> None: + """Trader-template path: the bare question serves as both values.""" + question, query, tier = parse_prompt(TRADER_PROMPT) + assert question == "Will X happen?" + assert query == question + assert tier == "template" + + def test_free_text_llm_gets_full_prompt_query_is_the_clause(self) -> None: + """Free-text input: whole prompt to the LLM, question clause to Serper.""" + question, query, tier = parse_prompt(LONG_FREE_TEXT_PROMPT) + assert question == LONG_FREE_TEXT_PROMPT + # "Please predict the following market: " is gone; the clause survives + # whole, including the deadline and the trailing '?'. + assert query.startswith("Will Alexander Isak") + assert query.endswith("?") + assert len(query) <= _MAX_SEARCH_QUERY_LEN + assert tier == "clause" + + def test_boilerplate_lead_in_does_not_anchor_the_query(self) -> None: + """Boilerplate lead-in must not anchor; the market question wins.""" + prompt = ( + "You are being asked to provide a probability estimate for a " + "prediction market question. Please respond with a JSON object. " + "Question: Will Bitcoin reach $150,000 or higher on any major " + "exchange by December 31, 2026? Resolution source: TradingView." + ) + _, query, tier = parse_prompt(prompt) + assert query.startswith("Will Bitcoin reach") + assert query.endswith("2026?") + assert tier == "clause" + + def test_double_quotes_are_stripped_from_query_only(self) -> None: + """Quoted spans become exact-match Serper terms; drop them from the query.""" + prompt = ( + 'Will any candle have a final "High" price >= 82000 in the window? ' + "Resolution source: Binance." + ) + question, query, _ = parse_prompt(prompt) + assert '"' not in query + assert "High" in query + assert '"High"' in question # the LLM still sees the exact wording + + def test_tier_is_reported(self) -> None: + """The tier tags template / clause / raw explicitly.""" + assert parse_prompt(TRADER_PROMPT)[2] == "template" + assert parse_prompt(FREE_TEXT_PROMPT)[2] == "clause" + assert parse_prompt("no question mark here at all")[2] == "raw" + + def test_scan_window_bounds_candidate_search(self) -> None: + """A clause past the scan window is not found; the LLM still gets all.""" + prompt = "x" * (3 * _MAX_SCAN_CHARS) + " Will X happen by 2027?" + question, _, tier = parse_prompt(prompt) + assert tier == "raw" + assert question == prompt + + +class TestEmptyRetrievalGuard: + """v3 returns a flagged null prediction on empty retrieval (issue #455).""" + + @pytest.mark.parametrize( + "degenerate", + [ + "", + " ", + "???", + '"""', + "\u201c\u201d\u2018\u2019", + ], + ) + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_degenerate_prompt_short_circuits_before_serper( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock, degenerate: str + ) -> None: + """Prompts with no searchable content never reach Serper at all.""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=degenerate, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_fetch.assert_not_called() + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["p_no"] == 0.5 + assert parsed["confidence"] == 0.0 and parsed["info_utility"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "empty query" + assert result[4]["scan_truncated"] is False + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_both_empty_live_search_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Empty organic AND peopleAlsoAsk yields the flagged null, no LLM call.""" + mock_fetch.return_value = MagicMock(json=lambda: EMPTY_SERPER_RESPONSE) + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["confidence"] == 0.0 + assert result[4]["empty_retrieval"] is True + assert result[4]["null_reason"] == "live search" + assert result[4]["parse_tier"] == "clause" + mock_client.completions.assert_not_called() + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_cached_replay_both_empty_returns_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """An empty cached source_content yields the flagged null, no fetch.""" + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + source_content={ + "mode": "cleaned", + "serper_response": EMPTY_SERPER_RESPONSE, + }, + ) + parsed = json.loads(result[0]) + assert parsed["p_yes"] == 0.5 and parsed["confidence"] == 0.0 + assert result[4]["null_reason"] == "cached replay" + mock_fetch.assert_not_called() + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_reshaped_serper_body_is_an_error_not_a_flagged_null( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """A 200 body without the organic key surfaces as an error, not 0.5.""" + mock_fetch.return_value = MagicMock(json=lambda: {"message": "quota exceeded"}) + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + # v3's decorator wraps unexpected exceptions as a stringified error + # tuple; the shape ValueError must be visible there, not a 0.5 null. + assert "organic" in result[0] + assert result[4] is None + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_organic_empty_but_misc_present_still_calls_llm( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """The guard needs BOTH lists empty; PAA alone keeps the LLM path.""" + mock_fetch.return_value = MagicMock( + json=lambda: { + "organic": [], + "peopleAlsoAsk": [{"question": "Q?", "snippet": "A."}], + } + ) + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + mock_client.completions.assert_called_once() + assert json.loads(result[0])["p_yes"] == 0.6 + + +class TestRunWiring: + """run() feeds parse_prompt's outputs to the right consumers.""" + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_trader_request_sends_extracted_question_to_serper( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """LLM-input parity: a trader request searches the bare question.""" + serper_resp = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_fetch.return_value = serper_resp + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=TRADER_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + # the HTTP-error guard must actually run on the happy path (a MagicMock + # would silently absorb its removal otherwise) + serper_resp.raise_for_status.assert_called_once() + query_sent = mock_fetch.call_args[0][0] + assert query_sent == "Will X happen?" + # and the LLM prompt carries the bare question, not the full template + llm_prompt = mock_client.completions.call_args.kwargs["messages"][1]["content"] + assert "Will X happen?" in llm_prompt + assert "`yes` answer criterion" not in llm_prompt + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_free_text_serper_gets_short_query_llm_gets_full_prompt( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Serper gets the derived clause; the LLM sees the whole prompt.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + mock_client = _install_mock_client(mock_client_mgr) + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=LONG_FREE_TEXT_PROMPT, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + query_sent = mock_fetch.call_args[0][0] + assert len(query_sent) <= _MAX_SEARCH_QUERY_LEN + assert query_sent != LONG_FREE_TEXT_PROMPT + # criteria text that the derived query drops must still reach the LLM + llm_prompt = mock_client.completions.call_args.kwargs["messages"][1]["content"] + assert "official club announcements or BBC Sport" in llm_prompt + assert result[1] == llm_prompt + assert result[4]["parse_tier"] == "clause" + assert result[4]["scan_truncated"] is False + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_long_template_prompt_is_not_marked_truncated( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Template past the window is NOT flagged: question precedes the scan.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + _install_mock_client(mock_client_mgr) + prompt = TRADER_PROMPT + " filler" * (_MAX_SCAN_CHARS // 3) + assert len(prompt) > _MAX_SCAN_CHARS + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=prompt, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + assert result[4]["parse_tier"] == "template" + assert result[4]["scan_truncated"] is False + + @patch(f"{V3_MODULE}.LLMClientManager") + @patch(f"{V3_MODULE}.fetch_additional_sources") + def test_scan_truncation_is_observable( + self, mock_fetch: MagicMock, mock_client_mgr: MagicMock + ) -> None: + """Raw tier from an exhausted scan window is marked, not silent.""" + mock_fetch.return_value = MagicMock(json=lambda: FAKE_SERPER_RESPONSE) + _install_mock_client(mock_client_mgr) + # the only '?' sits past the scan window -> raw tier via truncation + prompt = "word " * (_MAX_SCAN_CHARS // 4) + "Will it happen by 2027?" + result = run( + tool="superforcaster-polymarket-v3", + model="claude-fable-5", + prompt=prompt, + api_keys=_make_mock_api_keys(), + counter_callback=None, + ) + assert result[4]["parse_tier"] == "raw" + assert result[4]["scan_truncated"] is True diff --git a/packages/valory/services/mech_predict/service.yaml b/packages/valory/services/mech_predict/service.yaml index 1befec0a2..6481e6d3a 100644 --- a/packages/valory/services/mech_predict/service.yaml +++ b/packages/valory/services/mech_predict/service.yaml @@ -7,7 +7,7 @@ license: Apache-2.0 fingerprint: README.md: bafybeifogoiwjqlb3nk2o6bbybr4safi4bfv4va7a4rq7ptse6ps7ep4zm fingerprint_ignore_patterns: [] -agent: valory/mech_predict:0.1.0:bafybeigyc7wnb6hko7rc7y2kwnvnckjsck4uffhlmdclrvpyp2jww7scrm +agent: valory/mech_predict:0.1.0:bafybeibafk2js4wyidwhjpdtddcjfsitbusnfnof6pcdxkxuggmdjmgary number_of_agents: 1 deployment: agent: From 8a59cd9c0f2577a4cfadcca1b1f7c5370ede57c3 Mon Sep 17 00:00:00 2001 From: jmoreira-valory Date: Sat, 5 Sep 2026 18:33:06 +0200 Subject: [PATCH 2/2] fix: pylint suppressions that survive formatter line-wrapping CI pylint flagged the napthaai test files: black had re-wrapped the long assert lines, moving the code off the line-level disable comments. Replace them with module-level private-cap aliases (one disable each, below the imports) and the pylint-suggested falsiness assert. Co-Authored-By: Claude Fable 5 --- .../prediction_request_rag_v1/component.yaml | 2 +- .../tests/test_prediction_request_rag_v1.py | 19 +++++++++---------- .../component.yaml | 2 +- .../test_prediction_request_reasoning_v1.py | 15 ++++++++------- packages/packages.json | 8 ++++---- .../agents/mech_predict/aea-config.yaml | 4 ++-- .../valory/services/mech_predict/service.yaml | 2 +- 7 files changed, 26 insertions(+), 26 deletions(-) diff --git a/packages/napthaai/customs/prediction_request_rag_v1/component.yaml b/packages/napthaai/customs/prediction_request_rag_v1/component.yaml index 12685aae2..f296331d0 100644 --- a/packages/napthaai/customs/prediction_request_rag_v1/component.yaml +++ b/packages/napthaai/customs/prediction_request_rag_v1/component.yaml @@ -9,7 +9,7 @@ fingerprint: __init__.py: bafybeifnw5qoyshsiq2g7ulz4q3vbjpy7pxgxpop3f5dsrnfgm5w3dahz4 prediction_request_rag_v1.py: bafybeibr34vvrq2s4morz4tr4ahn7vfuc2gegtow6rkgs44jdphfd37sx4 tests/__init__.py: bafybeifcgilmgfwx7kaap67cfnuokrhfrabi6bnvqiudyzgf2idu64dvxq - tests/test_prediction_request_rag_v1.py: bafybeiayzev7seunarcbociixwqqqrajraoc3dflgesmlmgdr4o4tjp5rq + tests/test_prediction_request_rag_v1.py: bafybeidv3ivxgjtc7mvgwvorpuwxhu2eihrdwqnjfesp3unaioz5bge6ha fingerprint_ignore_patterns: [] entry_point: prediction_request_rag_v1.py callable: run diff --git a/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py b/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py index 4083a35cd..b8922632b 100644 --- a/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py +++ b/packages/napthaai/customs/prediction_request_rag_v1/tests/test_prediction_request_rag_v1.py @@ -41,6 +41,11 @@ run, ) +# Aliases for module-private caps: one disable each here, so call sites stay +# clean and the suppression cannot drift under formatter line-wrapping. +_QUERY_CAP = module._MAX_SEARCH_QUERY_LEN # pylint: disable=protected-access +_SCAN_CAP = module._MAX_SCAN_CHARS # pylint: disable=protected-access + class TestLLMClientManager: """Verify LLMClientManager creates per-context clients without globals.""" @@ -707,9 +712,7 @@ def test_boilerplate_prefix_is_dropped_from_query(self) -> None: _, query, _ = module.parse_prompt(LONG_FREE_TEXT_PROMPT) assert query.startswith("Will Alexander Isak") assert query.endswith("?") - assert ( - len(query) <= module._MAX_SEARCH_QUERY_LEN - ) # pylint: disable=protected-access + assert len(query) <= _QUERY_CAP class TestDegenerateShortCircuit: @@ -805,9 +808,7 @@ def test_empty_serper_body_returns_no_urls(self) -> None: serper_resp.raise_for_status.return_value = None serper_resp.json.return_value = {"organic": [], "peopleAlsoAsk": []} with patch(f"{RAG_MODULE}.requests.request", return_value=serper_resp): - assert ( - module.get_urls_from_queries_serper(["q"], api_key="k", num=5) == [] - ) # pylint: disable=use-implicit-booleaness-not-comparison + assert not module.get_urls_from_queries_serper(["q"], api_key="k", num=5) class TestRunParityAndParseMetadata: @@ -844,10 +845,8 @@ def test_trader_template_feeds_extracted_question_everywhere(self) -> None: def test_long_template_prompt_is_not_marked_truncated(self) -> None: """Template past the scan window is NOT flagged as truncated.""" - prompt = TRADER_PROMPT + " filler" * ( - module._MAX_SCAN_CHARS // 3 - ) # pylint: disable=protected-access - assert len(prompt) > module._MAX_SCAN_CHARS # pylint: disable=protected-access + prompt = TRADER_PROMPT + " filler" * (_SCAN_CAP // 3) + assert len(prompt) > _SCAN_CAP result, _ = self._run_with_fetch_mock(prompt) assert result[4]["parse_tier"] == "template" assert result[4]["scan_truncated"] is False diff --git a/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml b/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml index e8187cd3c..c62049158 100644 --- a/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml +++ b/packages/napthaai/customs/prediction_request_reasoning_v1/component.yaml @@ -10,7 +10,7 @@ fingerprint: __init__.py: bafybeibcbvmr7v5n2vintaunxjix2jxopbvjij5hytewh3xtlca4mfwhky prediction_request_reasoning_v1.py: bafybeifjrnrdx47s2n3bwggblf5bwa7cgkr3qq2emy2tgyx7twera6gmfu tests/__init__.py: bafybeieu3toasuyaehkqounrtylk5bjer7qgvnt6ccjcsi6jlfig7wfrka - tests/test_prediction_request_reasoning_v1.py: bafybeid4acfxihfnm6l7usi4sy7cjkgdnfclxny6u5ysyjsf2fqyfdiafq + tests/test_prediction_request_reasoning_v1.py: bafybeihvdjgc65v4nz66vmlciyrtqekl4sf2ng4g3bfgg2kiaf3bjwv4re fingerprint_ignore_patterns: [] entry_point: prediction_request_reasoning_v1.py callable: run diff --git a/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py b/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py index d55ffdeb7..b80d16e0b 100644 --- a/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py +++ b/packages/napthaai/customs/prediction_request_reasoning_v1/tests/test_prediction_request_reasoning_v1.py @@ -45,6 +45,11 @@ run, ) +# Aliases for module-private caps: one disable each here, so call sites stay +# clean and the suppression cannot drift under formatter line-wrapping. +_QUERY_CAP = module._MAX_SEARCH_QUERY_LEN # pylint: disable=protected-access +_SCAN_CAP = module._MAX_SCAN_CHARS # pylint: disable=protected-access + class TestLLMClientManager: """Verify LLMClientManager creates per-context clients without globals.""" @@ -712,9 +717,7 @@ def test_boilerplate_prefix_is_dropped_from_query(self) -> None: assert tier == "clause" assert query.startswith("Will Alexander Isak") assert query.endswith("?") - assert ( - len(query) <= module._MAX_SEARCH_QUERY_LEN - ) # pylint: disable=protected-access + assert len(query) <= _QUERY_CAP def test_tier_is_reported(self) -> None: """The tier tags template / clause / raw explicitly.""" @@ -913,10 +916,8 @@ def test_long_template_prompt_not_marked_truncated( content="0.5", usage=MagicMock(prompt_tokens=10, completion_tokens=5), ) - prompt = TRADER_PROMPT + " filler" * ( - module._MAX_SCAN_CHARS // 3 - ) # pylint: disable=protected-access - assert len(prompt) > module._MAX_SCAN_CHARS # pylint: disable=protected-access + prompt = TRADER_PROMPT + " filler" * (_SCAN_CAP // 3) + assert len(prompt) > _SCAN_CAP result = run( tool="prediction-request-reasoning-v1", model="gpt-4.1-2025-04-14", diff --git a/packages/packages.json b/packages/packages.json index 2c0e49399..180c38934 100644 --- a/packages/packages.json +++ b/packages/packages.json @@ -2,8 +2,8 @@ "dev": { "custom/dvilela/corcel_request/0.1.0": "bafybeic74xc32orfiffumv3cxe7o4vktzobk6gszjecuhto4or7k5ppi5y", "custom/dvilela/gemini_prediction/0.1.0": "bafybeifpoil2wjh6hr7sqoomlilhsyt3ktpy6tjzmisrlopxcjbkr3hhfq", - "custom/napthaai/prediction_request_rag_v1/0.1.0": "bafybeierk5iyyf3tlpjqtxi5unawhih2sd2pwkp7owaxwszrxvatecdbdy", - "custom/napthaai/prediction_request_reasoning_v1/0.1.0": "bafybeicr7zpvpbg6kktaqi4shxpyok4gboqdbuxyd25zcsof6yljlffaie", + "custom/napthaai/prediction_request_rag_v1/0.1.0": "bafybeictfszce2ftkcxrz5xkzwooqx44onb5ylicrswfdgm4westhg5dyi", + "custom/napthaai/prediction_request_reasoning_v1/0.1.0": "bafybeibvfj4v55wdlga2h6pmfylqfpdwlcgouldmfvyy2mudqsubycgrzi", "custom/napthaai/prediction_url_cot_v1/0.1.0": "bafybeife4x6syyji4jtfbg2jmxec7b2bv3z6d4lkvj6a7da5nhelrunyja", "custom/napthaai/resolve_market_reasoning/0.1.0": "bafybeidji6or6kpmawho64tc7qetinykhrklvv2gmnqfcd6ejmg43j3i7q", "custom/nickcom007/prediction_request_sme/0.1.0": "bafybeifsk4og24t22agcrj7mm6fz3wxwfbdg554yjhc6odpxal5ofg22ti", @@ -28,8 +28,8 @@ "custom/valory/superforcaster_polymarket_v4/0.1.0": "bafybeiefu5cetnebkr2yza6yn6la2ldkwvn2pjlgcmew2e6fblnyjh5roq", "custom/victorpolisetty/dalle_request/0.1.0": "bafybeiadatvfc6opcsmhpxxzmsqpwgbckblciqypx2iwe5uwove6nmki3e", "custom/victorpolisetty/gemini_request/0.1.0": "bafybeiamjk5mfycjqkstkzymggako2jt7yjros2zx4l54zyuei2xkj4gqu", - "agent/valory/mech_predict/0.1.0": "bafybeibafk2js4wyidwhjpdtddcjfsitbusnfnof6pcdxkxuggmdjmgary", - "service/valory/mech_predict/0.1.0": "bafybeiceslxfai4qdbire7natjjqbrlg4wclyye52elsopvg3sit5sb4vm" + "agent/valory/mech_predict/0.1.0": "bafybeie6wgehowtsrgnptosufbq25kfmocwbrmg7ykoacl5k2usifyyzh4", + "service/valory/mech_predict/0.1.0": "bafybeibxkw56nnmxipf2bydkdvyfwnnb2zoxzc2qjyb4646ww3mjotp5he" }, "third_party": { "protocol/open_aea/signing/1.0.0": "bafybeifsjmldwyki3beqyvdt5lzenrg6wyrqaar5plc5rpnvtc4zlentye", diff --git a/packages/valory/agents/mech_predict/aea-config.yaml b/packages/valory/agents/mech_predict/aea-config.yaml index 744022e5a..c2a46b7d3 100644 --- a/packages/valory/agents/mech_predict/aea-config.yaml +++ b/packages/valory/agents/mech_predict/aea-config.yaml @@ -56,8 +56,8 @@ customs: - valory/resolve_market_jury:0.1.0:bafybeigdv6zbbuf2flnti7hfwbotagnqz43g76v2jfhcuk4ancucxmg5gi - valory/prediction_request_v1:0.1.0:bafybeie23dpatwki6odlv7nnlwibuq6jbdonsmu5mszsgitlwghzpiesam - napthaai/resolve_market_reasoning:0.1.0:bafybeidji6or6kpmawho64tc7qetinykhrklvv2gmnqfcd6ejmg43j3i7q -- napthaai/prediction_request_rag_v1:0.1.0:bafybeierk5iyyf3tlpjqtxi5unawhih2sd2pwkp7owaxwszrxvatecdbdy -- napthaai/prediction_request_reasoning_v1:0.1.0:bafybeicr7zpvpbg6kktaqi4shxpyok4gboqdbuxyd25zcsof6yljlffaie +- napthaai/prediction_request_rag_v1:0.1.0:bafybeictfszce2ftkcxrz5xkzwooqx44onb5ylicrswfdgm4westhg5dyi +- napthaai/prediction_request_reasoning_v1:0.1.0:bafybeibvfj4v55wdlga2h6pmfylqfpdwlcgouldmfvyy2mudqsubycgrzi - valory/prepare_tx:0.1.0:bafybeidpvkwtn5m5yjp5mr2tn6f52btmlki32syahvkkuvlw3oq5eee3di - napthaai/prediction_url_cot_v1:0.1.0:bafybeife4x6syyji4jtfbg2jmxec7b2bv3z6d4lkvj6a7da5nhelrunyja - valory/prediction_langchain:0.1.0:bafybeieirig3irwmfubxjoxcujiwxxb7knl3c3amhqhjy7bxvo2tdvybpm diff --git a/packages/valory/services/mech_predict/service.yaml b/packages/valory/services/mech_predict/service.yaml index 6481e6d3a..5bde76d24 100644 --- a/packages/valory/services/mech_predict/service.yaml +++ b/packages/valory/services/mech_predict/service.yaml @@ -7,7 +7,7 @@ license: Apache-2.0 fingerprint: README.md: bafybeifogoiwjqlb3nk2o6bbybr4safi4bfv4va7a4rq7ptse6ps7ep4zm fingerprint_ignore_patterns: [] -agent: valory/mech_predict:0.1.0:bafybeibafk2js4wyidwhjpdtddcjfsitbusnfnof6pcdxkxuggmdjmgary +agent: valory/mech_predict:0.1.0:bafybeie6wgehowtsrgnptosufbq25kfmocwbrmg7ykoacl5k2usifyyzh4 number_of_agents: 1 deployment: agent: