Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 32 additions & 2 deletions quantmind/preprocess/news.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,9 @@
r"(?<!\*)\*{1,2}([A-Z][A-Z0-9.-]{0,9})\*{1,2}(?!\*)",
re.IGNORECASE,
)
_SHARED_EXCHANGE_SYMBOL_RE = re.compile(
r"\s*,\s*([A-Z][A-Z0-9.-]{0,9})",
)
_EMAIL_PROTECTION_LINK_RE = re.compile(
r"\[\[email protected]\]\(/cdn-cgi/l/email-protection#[^)]+\)"
)
Expand Down Expand Up @@ -401,8 +404,9 @@ def canonicalize_source_url(url: str) -> str:
def extract_exchange_ticker_hints(text: str) -> tuple[NewsTickerHint, ...]:
"""Extract exchange-qualified ticker mentions from PR-style text.

Examples matched include ``(NASDAQ: NVDA)`` and ``NYSE: IBM``. The result is
only a hint; downstream instrument resolution should still validate it.
Examples matched include ``(NASDAQ: NVDA)``, ``NYSE: IBM``, and a
parenthesized comma list such as ``(NYSE: EVEX, EVEXW)``. The result is only
a hint; downstream instrument resolution should still validate it.
Markdown link and emphasis decoration is removed from a scan-only copy so
stored news text and its content hash retain their original representation.
"""
Expand All @@ -425,6 +429,32 @@ def extract_exchange_ticker_hints(text: str) -> tuple[NewsTickerHint, ...]:
raw=match.group(0).strip(),
)
)
matched_text = match.group(0)
if not matched_text.lstrip().startswith(
"("
) or matched_text.rstrip().endswith(")"):
continue

closing_parenthesis = scan_text.find(")", match.end())
if closing_parenthesis == -1:
continue
suffix = scan_text[match.end() : closing_parenthesis]
position = 0
while continuation := _SHARED_EXCHANGE_SYMBOL_RE.match(
suffix, position
):
symbol = continuation.group(1).upper()
key = (symbol, exchange)
if key not in seen:
seen.add(key)
hints.append(
NewsTickerHint(
symbol=symbol,
exchange=exchange,
raw=continuation.group(0).strip(),
)
)
position = continuation.end()
return tuple(hints)


Expand Down
52 changes: 52 additions & 0 deletions tests/preprocess/test_news.py
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,58 @@ def test_exchange_ticker_hints_ignore_markdown_decoration(self):
expected,
)

def test_exchange_ticker_hints_capture_shared_prefix_comma_list(self):
hints = extract_exchange_ticker_hints(
"Example issuer (NYSE: EVEX, EVEXW) announced results."
)

self.assertEqual(
tuple((hint.symbol, hint.exchange, hint.raw) for hint in hints),
(
("EVEX", "NYSE", "(NYSE: EVEX"),
("EVEXW", "NYSE", ", EVEXW"),
),
)

def test_exchange_ticker_hints_stop_at_shared_prefix_boundaries(self):
cases = (
(
"balanced mention followed by prose",
"Shares of (NYSE: IBM), and (NASDAQ: AAPL) rose today.",
(
("IBM", "NYSE", "(NYSE: IBM)"),
("AAPL", "NASDAQ", "(NASDAQ: AAPL)"),
),
),
("unsupported exchange", "(OTCID: QVCAQ, QVCGQ, QVCPQ)", ()),
(
"conjunction boundary",
"(NYSE: TME and HKEX: 1698)",
(("TME", "NYSE", "(NYSE: TME"),),
),
(
"semicolon boundary TSXV",
"(NASDAQ: VMAR; TSXV: VMAR)",
(("VMAR", "NASDAQ", "(NASDAQ: VMAR"),),
),
(
"semicolon boundary BMV",
"(NYSE: ASR; BMV: ASUR)",
(("ASR", "NYSE", "(NYSE: ASR"),),
),
)

for name, text, expected in cases:
with self.subTest(name=name):
hints = extract_exchange_ticker_hints(text)

self.assertEqual(
tuple(
(hint.symbol, hint.exchange, hint.raw) for hint in hints
),
expected,
)

def test_build_sec_news_identity(self):
self.assertEqual(
build_sec_news_identity(
Expand Down