diff --git a/quantmind/preprocess/news.py b/quantmind/preprocess/news.py index 12e7706..428268f 100644 --- a/quantmind/preprocess/news.py +++ b/quantmind/preprocess/news.py @@ -56,6 +56,9 @@ r"(? str: def extract_exchange_ticker_hints(text: str) -> tuple[NewsTickerHint, ...]: """Extract exchange-qualified ticker mentions from PR-style text. - Examples matched include ``(NASDAQ: NVDA)`` and ``NYSE: IBM``. The result is - only a hint; downstream instrument resolution should still validate it. + Examples matched include ``(NASDAQ: NVDA)``, ``NYSE: IBM``, and a + parenthesized comma list such as ``(NYSE: EVEX, EVEXW)``. The result is only + a hint; downstream instrument resolution should still validate it. Markdown link and emphasis decoration is removed from a scan-only copy so stored news text and its content hash retain their original representation. """ @@ -425,6 +429,32 @@ def extract_exchange_ticker_hints(text: str) -> tuple[NewsTickerHint, ...]: raw=match.group(0).strip(), ) ) + matched_text = match.group(0) + if not matched_text.lstrip().startswith( + "(" + ) or matched_text.rstrip().endswith(")"): + continue + + closing_parenthesis = scan_text.find(")", match.end()) + if closing_parenthesis == -1: + continue + suffix = scan_text[match.end() : closing_parenthesis] + position = 0 + while continuation := _SHARED_EXCHANGE_SYMBOL_RE.match( + suffix, position + ): + symbol = continuation.group(1).upper() + key = (symbol, exchange) + if key not in seen: + seen.add(key) + hints.append( + NewsTickerHint( + symbol=symbol, + exchange=exchange, + raw=continuation.group(0).strip(), + ) + ) + position = continuation.end() return tuple(hints) diff --git a/tests/preprocess/test_news.py b/tests/preprocess/test_news.py index 7a0b634..d8745f0 100644 --- a/tests/preprocess/test_news.py +++ b/tests/preprocess/test_news.py @@ -117,6 +117,58 @@ def test_exchange_ticker_hints_ignore_markdown_decoration(self): expected, ) + def test_exchange_ticker_hints_capture_shared_prefix_comma_list(self): + hints = extract_exchange_ticker_hints( + "Example issuer (NYSE: EVEX, EVEXW) announced results." + ) + + self.assertEqual( + tuple((hint.symbol, hint.exchange, hint.raw) for hint in hints), + ( + ("EVEX", "NYSE", "(NYSE: EVEX"), + ("EVEXW", "NYSE", ", EVEXW"), + ), + ) + + def test_exchange_ticker_hints_stop_at_shared_prefix_boundaries(self): + cases = ( + ( + "balanced mention followed by prose", + "Shares of (NYSE: IBM), and (NASDAQ: AAPL) rose today.", + ( + ("IBM", "NYSE", "(NYSE: IBM)"), + ("AAPL", "NASDAQ", "(NASDAQ: AAPL)"), + ), + ), + ("unsupported exchange", "(OTCID: QVCAQ, QVCGQ, QVCPQ)", ()), + ( + "conjunction boundary", + "(NYSE: TME and HKEX: 1698)", + (("TME", "NYSE", "(NYSE: TME"),), + ), + ( + "semicolon boundary TSXV", + "(NASDAQ: VMAR; TSXV: VMAR)", + (("VMAR", "NASDAQ", "(NASDAQ: VMAR"),), + ), + ( + "semicolon boundary BMV", + "(NYSE: ASR; BMV: ASUR)", + (("ASR", "NYSE", "(NYSE: ASR"),), + ), + ) + + for name, text, expected in cases: + with self.subTest(name=name): + hints = extract_exchange_ticker_hints(text) + + self.assertEqual( + tuple( + (hint.symbol, hint.exchange, hint.raw) for hint in hints + ), + expected, + ) + def test_build_sec_news_identity(self): self.assertEqual( build_sec_news_identity(