"""Deterministic financial parser using regex-based detection. Detects tickers, currencies, money amounts, percentages, basis points, ranges, EPS, revenue, dates, and fiscal periods from source text. Each match returns exact character offsets into the source text. """ from __future__ import annotations import re from services.intelligence_pipeline_v3.parsing.models import ( CandidateType, ParsedCandidate, PeriodAnnotation, ) from services.intelligence_pipeline_v3.parsing.normalizer import ( normalize_basis_points, normalize_money, normalize_percentage, normalize_range, ) # --------------------------------------------------------------------------- # Regex patterns # --------------------------------------------------------------------------- # Ticker: $AAPL or standalone AAPL-like (1-5 uppercase letters) _TICKER_DOLLAR_RE = re.compile(r"\$([A-Z]{1,5})\b") _TICKER_BARE_RE = re.compile(r"\b([A-Z]{1,5})\b") # Common English words that look like tickers but aren't _TICKER_STOPWORDS = frozenset({ "A", "I", "AM", "AN", "AS", "AT", "BE", "BY", "DO", "GO", "HE", "IF", "IN", "IS", "IT", "ME", "MY", "NO", "OF", "OK", "ON", "OR", "OUR", "SO", "THE", "TO", "UP", "US", "WE", "CEO", "CFO", "COO", "CTO", "EPS", "ETF", "GDP", "IPO", "LLC", "LTD", "NYSE", "SEC", "USA", "AND", "ARE", "BUT", "CAN", "DID", "FOR", "GET", "GOT", "HAD", "HAS", "HER", "HIS", "HOW", "ITS", "LET", "MAY", "NEW", "NOT", "NOW", "OLD", "OUR", "OWN", "PUT", "RAN", "SAY", "SHE", "TOO", "TWO", "USE", "WAS", "WAY", "WHO", "WHY", "WIN", "WON", "YET", "YOU", "ALL", "ANY", "BIG", "DAY", "END", "FEW", "FAR", "HIT", "LOW", "MET", "NET", "OUT", "RUN", "SET", "TOP", "TRY", "ALSO", "BACK", "BEEN", "BEST", "BOTH", "CAME", "COME", "DOWN", "EACH", "FROM", "GAVE", "GOOD", "HAVE", "HERE", "HIGH", "INTO", "JUST", "KEEP", "LAST", "LONG", "MADE", "MAKE", "MANY", "MORE", "MOST", "MUCH", "MUST", "NEED", "NEXT", "ONLY", "OVER", "SAID", "SAME", "SOME", "SUCH", "TAKE", "THAN", "THAT", "THEM", "THEN", "THEY", "THIS", "VERY", "WANT", "WELL", "WENT", "WERE", "WHAT", "WHEN", "WILL", "WITH", "WORK", "YEAR", "YOUR", "ITEM", "CASH", "FLOW", "FREE", "FULL", "HALF", "RISE", "ROSE", "FELL", "BEAT", "MISS", "GREW", "GROW", "LOST", "LOSS", "GAIN", "HOLD", "SELL", "CALL", "BUY", "FUND", "BOND", "RATE", "DEBT", "DEAL", "RISK", "Q", "H", "FY", "YOY", "QOQ", "AI", "R", "D", }) # Fiscal period: Q1 2024, Q4'24, FY2025, FY25, H1 2024 _FISCAL_PERIOD_RE = re.compile( r"\b(Q[1-4]|H[12]|FY)\s*['\u2019]?\s*(\d{4}|\d{2})\b" ) # Date patterns: January 15, 2024 / Jan 15, 2024 / 2024-01-15 _MONTH_NAMES = ( r"(?:January|February|March|April|May|June|July|August|September|October|November|December" r"|Jan|Feb|Mar|Apr|Jun|Jul|Aug|Sep|Oct|Nov|Dec)" ) _DATE_NAMED_RE = re.compile( rf"\b({_MONTH_NAMES})\s+(\d{{1,2}})(?:,?\s+(\d{{4}}))?\b" ) _DATE_ISO_RE = re.compile(r"\b(\d{4})-(\d{2})-(\d{2})\b") # Basis points: "25 basis points", "50bps", "25 bps" _BASIS_POINTS_RE = re.compile( r"[+-]?\d[\d,]*\.?\d*\s*(?:basis\s+points?|bps)\b", re.IGNORECASE ) # Percentage: 4%, -2.5%, +1.2 percent _PERCENTAGE_RE = re.compile( r"[+-]?\d[\d,]*\.?\d*\s*(?:%|percent(?:age)?(?:\s+points?)?\b)", re.IGNORECASE ) # EPS: "$1.52 per share", "earnings per share of $1.52" _EPS_PER_SHARE_RE = re.compile( r"\$\s*\d[\d,]*\.?\d*\s+per\s+share\b", re.IGNORECASE ) _EPS_PREFIX_RE = re.compile( r"\b(?:EPS|earnings\s+per\s+share)\s+(?:of\s+)?\$\s*\d[\d,]*\.?\d*", re.IGNORECASE ) # Revenue: "$94.9 billion in revenue", "revenue of $94.9 billion" _REVENUE_AMOUNT_RE = re.compile( r"\$\s*\d[\d,]*\.?\d*\s*(?:trillion|billion|million|bn|mn)\s+(?:in\s+)?revenue\b", re.IGNORECASE, ) _REVENUE_PREFIX_RE = re.compile( r"\brevenue\s+(?:of|was|reached|hit|grew\s+to|increased\s+to|totaled)\s+\$\s*\d[\d,]*\.?\d*\s*(?:trillion|billion|million|bn|mn)?", re.IGNORECASE, ) # Range: "$10-$12", "$10 to $12", "$1.50-$2.00" _RANGE_RE = re.compile( r"\$\s*\d[\d,]*\.?\d*\s*(?:-|to|–|—)\s*\$?\s*\d[\d,]*\.?\d*", re.IGNORECASE, ) # Money with multiplier: "$94.9 billion", "$1.2 million", "€5 billion" _MONEY_MULT_RE = re.compile( r"[€£¥$]\s*\d[\d,]*\.?\d*\s*(?:trillion|billion|million|thousand|bn|mn|tn|[kmbt])\b", re.IGNORECASE, ) # Simple currency: $123.45, €99, £1,234.56 # Note: requires digits after decimal point to avoid matching trailing periods _CURRENCY_RE = re.compile( r"[€£¥$]\s*\d[\d,]*(?:\.\d+)?" ) # Currency codes: USD, EUR, GBP, JPY _CURRENCY_SYMBOLS = {"$": "USD", "€": "EUR", "£": "GBP", "¥": "JPY"} def _detect_currency_unit(text: str) -> str: """Detect currency unit from symbol in text.""" for symbol, code in _CURRENCY_SYMBOLS.items(): if symbol in text: return code return "USD" class FinancialParser: """Deterministic regex-based financial entity parser. Detects financial entities in text and returns ParsedCandidate instances with exact character offsets, literal text, and normalized values. """ def parse(self, text: str) -> list[ParsedCandidate]: """Parse text for financial entities. Returns a list of ParsedCandidate sorted by start_char offset. Overlapping matches are resolved by priority (more specific wins). """ if not text or not text.strip(): return [] candidates: list[ParsedCandidate] = [] # Order matters: more specific patterns first to claim offsets candidates.extend(self._parse_fiscal_periods(text)) candidates.extend(self._parse_dates(text)) candidates.extend(self._parse_basis_points(text)) candidates.extend(self._parse_eps(text)) candidates.extend(self._parse_revenue(text)) candidates.extend(self._parse_ranges(text)) candidates.extend(self._parse_percentages(text)) candidates.extend(self._parse_money_with_multiplier(text)) candidates.extend(self._parse_tickers(text)) candidates.extend(self._parse_currency(text)) # Resolve overlaps: keep higher-priority (earlier in list) matches candidates = self._resolve_overlaps(candidates) # Sort by position candidates.sort(key=lambda c: (c.start_char, -c.end_char)) return candidates def _resolve_overlaps(self, candidates: list[ParsedCandidate]) -> list[ParsedCandidate]: """Remove overlapping candidates, keeping earlier ones (higher priority).""" if not candidates: return [] # Sort by start position for greedy non-overlap resolution sorted_candidates = sorted(candidates, key=lambda c: (c.start_char, -c.end_char)) result: list[ParsedCandidate] = [] claimed: list[tuple[int, int]] = [] for cand in sorted_candidates: overlaps = False for start, end in claimed: # Check if this candidate overlaps with any claimed range if cand.start_char < end and cand.end_char > start: overlaps = True break if not overlaps: result.append(cand) claimed.append((cand.start_char, cand.end_char)) return result def _parse_tickers(self, text: str) -> list[ParsedCandidate]: """Parse ticker symbols: $AAPL style.""" results: list[ParsedCandidate] = [] # Dollar-prefixed tickers: $AAPL for match in _TICKER_DOLLAR_RE.finditer(text): ticker = match.group(1) if ticker not in _TICKER_STOPWORDS: results.append(ParsedCandidate( candidate_type=CandidateType.TICKER, literal_value=match.group(0), normalized_value=None, unit=None, start_char=match.start(), end_char=match.end(), )) return results def _parse_fiscal_periods(self, text: str) -> list[ParsedCandidate]: """Parse fiscal periods: Q1 2024, FY2025, H1 2024.""" results: list[ParsedCandidate] = [] for match in _FISCAL_PERIOD_RE.finditer(text): period_prefix = match.group(1) year_str = match.group(2) year = int(year_str) if len(year_str) == 2: year = 2000 + year if year < 80 else 1900 + year if period_prefix.startswith("Q"): period_type = "quarter" period_value = period_prefix elif period_prefix.startswith("H"): period_type = "half" period_value = period_prefix else: # FY period_type = "fiscal_year" period_value = "FY" results.append(ParsedCandidate( candidate_type=CandidateType.FISCAL_PERIOD, literal_value=match.group(0), normalized_value=None, unit=None, start_char=match.start(), end_char=match.end(), period=PeriodAnnotation( period_type=period_type, period_value=period_value, year=year, ), )) return results def _parse_dates(self, text: str) -> list[ParsedCandidate]: """Parse dates: January 15, 2024 / 2024-01-15.""" results: list[ParsedCandidate] = [] for match in _DATE_NAMED_RE.finditer(text): results.append(ParsedCandidate( candidate_type=CandidateType.DATE, literal_value=match.group(0), normalized_value=None, unit=None, start_char=match.start(), end_char=match.end(), )) for match in _DATE_ISO_RE.finditer(text): results.append(ParsedCandidate( candidate_type=CandidateType.DATE, literal_value=match.group(0), normalized_value=None, unit=None, start_char=match.start(), end_char=match.end(), )) return results def _parse_basis_points(self, text: str) -> list[ParsedCandidate]: """Parse basis points: 25 basis points, 50bps.""" results: list[ParsedCandidate] = [] for match in _BASIS_POINTS_RE.finditer(text): literal = match.group(0) normalized = normalize_basis_points(literal) results.append(ParsedCandidate( candidate_type=CandidateType.BASIS_POINTS, literal_value=literal, normalized_value=normalized, unit="bps", start_char=match.start(), end_char=match.end(), )) return results def _parse_percentages(self, text: str) -> list[ParsedCandidate]: """Parse percentages: 4%, -2.5%, +1.2 percent.""" results: list[ParsedCandidate] = [] for match in _PERCENTAGE_RE.finditer(text): literal = match.group(0) normalized = normalize_percentage(literal) results.append(ParsedCandidate( candidate_type=CandidateType.PERCENTAGE, literal_value=literal, normalized_value=normalized, unit="%", start_char=match.start(), end_char=match.end(), )) return results def _parse_eps(self, text: str) -> list[ParsedCandidate]: """Parse EPS values: $1.52 per share, EPS of $1.52.""" results: list[ParsedCandidate] = [] for match in _EPS_PER_SHARE_RE.finditer(text): literal = match.group(0) normalized = normalize_money(literal) results.append(ParsedCandidate( candidate_type=CandidateType.EPS, literal_value=literal, normalized_value=normalized, unit="USD", start_char=match.start(), end_char=match.end(), )) for match in _EPS_PREFIX_RE.finditer(text): literal = match.group(0) normalized = normalize_money(literal) results.append(ParsedCandidate( candidate_type=CandidateType.EPS, literal_value=literal, normalized_value=normalized, unit="USD", start_char=match.start(), end_char=match.end(), )) return results def _parse_revenue(self, text: str) -> list[ParsedCandidate]: """Parse revenue figures: $94.9 billion in revenue, revenue of $50 billion.""" results: list[ParsedCandidate] = [] for match in _REVENUE_AMOUNT_RE.finditer(text): literal = match.group(0) normalized = normalize_money(literal) results.append(ParsedCandidate( candidate_type=CandidateType.REVENUE, literal_value=literal, normalized_value=normalized, unit="USD", start_char=match.start(), end_char=match.end(), )) for match in _REVENUE_PREFIX_RE.finditer(text): literal = match.group(0) normalized = normalize_money(literal) results.append(ParsedCandidate( candidate_type=CandidateType.REVENUE, literal_value=literal, normalized_value=normalized, unit="USD", start_char=match.start(), end_char=match.end(), )) return results def _parse_ranges(self, text: str) -> list[ParsedCandidate]: """Parse ranges: $10-$12, $1.50 to $2.00.""" results: list[ParsedCandidate] = [] for match in _RANGE_RE.finditer(text): literal = match.group(0) low, high = normalize_range(literal) # Store midpoint as normalized value normalized = None if low is not None and high is not None: normalized = (low + high) / 2.0 unit = _detect_currency_unit(literal) results.append(ParsedCandidate( candidate_type=CandidateType.RANGE, literal_value=literal, normalized_value=normalized, unit=unit, start_char=match.start(), end_char=match.end(), )) return results def _parse_money_with_multiplier(self, text: str) -> list[ParsedCandidate]: """Parse money with multiplier: $94.9 billion, €5 million.""" results: list[ParsedCandidate] = [] for match in _MONEY_MULT_RE.finditer(text): literal = match.group(0) normalized = normalize_money(literal) unit = _detect_currency_unit(literal) results.append(ParsedCandidate( candidate_type=CandidateType.MONEY, literal_value=literal, normalized_value=normalized, unit=unit, start_char=match.start(), end_char=match.end(), )) return results def _parse_currency(self, text: str) -> list[ParsedCandidate]: """Parse simple currency: $123.45, €99, £1,234.56.""" results: list[ParsedCandidate] = [] for match in _CURRENCY_RE.finditer(text): literal = match.group(0) normalized = normalize_money(literal) unit = _detect_currency_unit(literal) results.append(ParsedCandidate( candidate_type=CandidateType.CURRENCY, literal_value=literal, normalized_value=normalized, unit=unit, start_char=match.start(), end_char=match.end(), )) return results