Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
59 changes: 48 additions & 11 deletions quantmind/preprocess/news.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,11 @@
r"(?<!\*)\*{1,2}([A-Z][A-Z0-9.-]{0,9})\*{1,2}(?!\*)",
re.IGNORECASE,
)
_SHARED_EXCHANGE_GROUP_RE = re.compile(
r"(?P<members>(?:\s*,\s*[A-Z][A-Z0-9.-]{0,9})+)"
r"(?=\s*(?:\)|;\s*[A-Z][A-Z0-9 .-]*\s*:))",
)
_SHARED_EXCHANGE_SYMBOL_RE = re.compile(r",\s*([A-Z][A-Z0-9.-]{0,9})")
_EMAIL_PROTECTION_LINK_RE = re.compile(
r"\[\[email protected]\]\(/cdn-cgi/l/email-protection#[^)]+\)"
)
Expand Down Expand Up @@ -401,8 +406,9 @@ def canonicalize_source_url(url: str) -> str:
def extract_exchange_ticker_hints(text: str) -> tuple[NewsTickerHint, ...]:
"""Extract exchange-qualified ticker mentions from PR-style text.

Examples matched include ``(NASDAQ: NVDA)`` and ``NYSE: IBM``. The result is
only a hint; downstream instrument resolution should still validate it.
Examples matched include ``(NASDAQ: NVDA)``, ``NYSE: IBM``, and a
parenthesized comma list such as ``(NYSE: EVEX, EVEXW)``. The result is only
a hint; downstream instrument resolution should still validate it.
Markdown link and emphasis decoration is removed from a scan-only copy so
stored news text and its content hash retain their original representation.
"""
Expand All @@ -414,17 +420,48 @@ def extract_exchange_ticker_hints(text: str) -> tuple[NewsTickerHint, ...]:
raw_exchange = " ".join(match.group(1).upper().split())
exchange = _EXCHANGE_NAMES.get(raw_exchange, raw_exchange)
symbol = match.group(2).upper()
continuation_matches: list[re.Match[str]] = []
shared_group_raw: str | None = None
matched_text = match.group(0)
if matched_text.lstrip().startswith(
"("
) and not matched_text.rstrip().endswith(")"):
closing_parenthesis = scan_text.find(")", match.end())
if closing_parenthesis != -1:
suffix = scan_text[match.end() : closing_parenthesis + 1]
shared_group = _SHARED_EXCHANGE_GROUP_RE.match(suffix)
if shared_group:
continuation_matches = list(
_SHARED_EXCHANGE_SYMBOL_RE.finditer(
shared_group.group("members")
)
)
shared_group_raw = scan_text[
match.start() : closing_parenthesis + 1
].strip()

key = (symbol, exchange)
if key in seen:
continue
seen.add(key)
hints.append(
NewsTickerHint(
symbol=symbol,
exchange=exchange,
raw=match.group(0).strip(),
if key not in seen:
seen.add(key)
hints.append(
NewsTickerHint(
symbol=symbol,
exchange=exchange,
raw=shared_group_raw or matched_text.strip(),
)
)
)
for continuation in continuation_matches:
symbol = continuation.group(1).upper()
key = (symbol, exchange)
if key not in seen:
seen.add(key)
hints.append(
NewsTickerHint(
symbol=symbol,
exchange=exchange,
raw=shared_group_raw,
)
)
return tuple(hints)


Expand Down
101 changes: 101 additions & 0 deletions tests/preprocess/test_news.py
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,107 @@ def test_exchange_ticker_hints_ignore_markdown_decoration(self):
expected,
)

def test_exchange_ticker_hints_capture_shared_prefix_comma_list(self):
hints = extract_exchange_ticker_hints(
"Example issuer (NYSE: EVEX, EVEXW) announced results."
)

self.assertEqual(
tuple((hint.symbol, hint.exchange, hint.raw) for hint in hints),
(
("EVEX", "NYSE", "(NYSE: EVEX, EVEXW)"),
("EVEXW", "NYSE", "(NYSE: EVEX, EVEXW)"),
),
)

def test_exchange_ticker_hints_stop_shared_list_at_semicolon(self):
hints = extract_exchange_ticker_hints(
"Example issuer (NYSE: EVEX, EVEXW; B3: EVEB31) announced results."
)

self.assertEqual(
tuple((hint.symbol, hint.exchange, hint.raw) for hint in hints),
(
(
"EVEX",
"NYSE",
"(NYSE: EVEX, EVEXW; B3: EVEB31)",
),
(
"EVEXW",
"NYSE",
"(NYSE: EVEX, EVEXW; B3: EVEB31)",
),
),
)

def test_exchange_ticker_hints_capture_new_shared_list_members(self):
hints = extract_exchange_ticker_hints(
"Prior (NYSE: EVEX). Offering (NYSE: EVEX, EVEXW)."
)

self.assertEqual(
tuple((hint.symbol, hint.exchange, hint.raw) for hint in hints),
(
("EVEX", "NYSE", "(NYSE: EVEX)"),
("EVEXW", "NYSE", "(NYSE: EVEX, EVEXW)"),
),
)

def test_exchange_ticker_hints_stop_at_shared_prefix_boundaries(self):
cases = (
(
"balanced mention followed by prose",
"Shares of (NYSE: IBM), and (NASDAQ: AAPL) rose today.",
(
("IBM", "NYSE", "(NYSE: IBM)"),
("AAPL", "NASDAQ", "(NASDAQ: AAPL)"),
),
),
("unsupported exchange", "(OTCID: QVCAQ, QVCGQ, QVCPQ)", ()),
(
"conjunction boundary",
"(NYSE: TME and HKEX: 1698)",
(("TME", "NYSE", "(NYSE: TME"),),
),
(
"semicolon boundary TSXV",
"(NASDAQ: VMAR; TSXV: VMAR)",
(("VMAR", "NASDAQ", "(NASDAQ: VMAR"),),
),
(
"semicolon boundary BMV",
"(NYSE: ASR; BMV: ASUR)",
(("ASR", "NYSE", "(NYSE: ASR"),),
),
(
"ordinary prose inside parentheses",
"(NYSE: IBM, and revenue increased)",
(("IBM", "NYSE", "(NYSE: IBM"),),
),
(
"uppercase token before ordinary prose",
"(NYSE: IBM, ABC and revenue increased)",
(("IBM", "NYSE", "(NYSE: IBM"),),
),
(
"uppercase token before non-exchange semicolon",
"(NYSE: IBM, ABC; revenue increased)",
(("IBM", "NYSE", "(NYSE: IBM"),),
),
)

for name, text, expected in cases:
with self.subTest(name=name):
hints = extract_exchange_ticker_hints(text)

self.assertEqual(
tuple(
(hint.symbol, hint.exchange, hint.raw) for hint in hints
),
expected,
)

def test_build_sec_news_identity(self):
self.assertEqual(
build_sec_news_identity(
Expand Down