Alternative Data for Alpha: Beyond Price and Volume
Discover how to extract trading signals from satellite imagery, social media, web scraping, and other non-traditional data sources in Indian markets.
Price and volume data are commodities—everyone has access. True alpha comes from information asymmetry. Alternative data sources—from satellite imagery tracking parking lots to social media sentiment—provide edges that traditional data cannot.
This practical guide shows you how to collect, process, and extract trading signals from alternative data sources accessible to Indian retail traders.
What is Alternative Data?
Traditional vs Alternative
Traditional Data:
- Price/volume from exchanges
- Financial statements
- Analyst reports
- Economic indicators
Alternative Data:
- Social media sentiment
- Web traffic analytics
- Satellite imagery
- Credit card transaction data
- Weather patterns
- Supply chain tracking
Legal Considerations in India
⚠️ Before collecting alternative data:
✅ Allowed:
- Public social media (Twitter/X, Reddit)
- Publicly accessible websites
- Government open data
- Satellite imagery (commercial providers)
❌ Not Allowed:
- Insider information
- Hacked/leaked data
- Data violating privacy laws
- Broker’s proprietary client data
SEBI Compliance: Alternative data strategies must be registered and audited like traditional algos.
Part 1: Web Scraping for Earnings Signals
Scraping Company Announcements
import requests
from bs4 import BeautifulSoup
import pandas as pd
from datetime import datetime, timedelta
from typing import List, Dict
import time
class BSEAnnouncementScraper:
"""
Scrape corporate announcements from BSE India
Strategy: Early detection of material events
(board meetings, results, dividends, acquisitions)
"""
def __init__(self):
self.base_url = "https://www.bseindia.com"
self.headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}
def get_corporate_announcements(self, scrip_code: str, days: int = 7) -> List[Dict]:
"""
Get recent announcements for a stock
Args:
scrip_code: BSE scrip code (e.g., '500325' for Reliance)
days: Number of days to look back
"""
url = f"{self.base_url}/corporates/ann.aspx?scripcd={scrip_code}&myDate={days}"
try:
response = requests.get(url, headers=self.headers, timeout=10)
response.raise_for_status()
soup = BeautifulSoup(response.content, 'html.parser')
announcements = []
# Parse announcement table
table = soup.find('table', {'class': 'TTRow'})
if table:
rows = table.find_all('tr')[1:] # Skip header
for row in rows:
cols = row.find_all('td')
if len(cols) >= 4:
date_str = cols[0].get_text().strip()
announcement = cols[1].get_text().strip()
category = cols[2].get_text().strip()
announcements.append({
'date': datetime.strptime(date_str, '%d %b %y'),
'announcement': announcement,
'category': category,
'scrip_code': scrip_code,
'source': 'BSE'
})
return announcements
except Exception as e:
print(f"Error scraping announcements: {e}")
return []
def classify_announcement(self, announcement: str) -> str:
"""
Classify announcement importance
Returns: 'HIGH', 'MEDIUM', 'LOW'
"""
high_keywords = [
'acquisition', 'merger', 'dividend', 'buyback',
'results', 'earnings', 'guidance', 'stake sale'
]
medium_keywords = [
'board meeting', 'agm', 'closure', 'record date'
]
announcement_lower = announcement.lower()
if any(keyword in announcement_lower for keyword in high_keywords):
return 'HIGH'
elif any(keyword in announcement_lower for keyword in medium_keywords):
return 'MEDIUM'
else:
return 'LOW'
def generate_signals(self, scrip_code: str) -> Dict:
"""
Generate trading signals from announcements
"""
announcements = self.get_corporate_announcements(scrip_code, days=7)
if not announcements:
return {'signal': 'NEUTRAL', 'reason': 'No announcements'}
# Classify announcements
for ann in announcements:
ann['importance'] = self.classify_announcement(ann['announcement'])
# Count high-priority announcements
high_priority = [a for a in announcements if a['importance'] == 'HIGH']
signal = {
'scrip_code': scrip_code,
'total_announcements': len(announcements),
'high_priority': len(high_priority),
'recent_announcements': announcements[:5],
'signal': 'NEUTRAL'
}
# Generate signal
if high_priority:
# Check sentiment of recent high-priority announcements
positive_keywords = ['dividend', 'buyback', 'acquisition', 'profit']
recent_high = high_priority[0]['announcement'].lower()
if any(kw in recent_high for kw in positive_keywords):
signal['signal'] = 'BUY'
signal['reason'] = f"Positive announcement: {high_priority[0]['announcement'][:100]}"
else:
signal['signal'] = 'WATCH'
signal['reason'] = f"Material announcement: {high_priority[0]['announcement'][:100]}"
return signal
# Example usage
scraper = BSEAnnouncementScraper()
# Monitor Reliance
signal = scraper.generate_signals('500325')
print(f"\n📢 Announcement Analysis: Reliance")
print(f" Signal: {signal['signal']}")
print(f" Total announcements (7 days): {signal['total_announcements']}")
print(f" High priority: {signal['high_priority']}")
if 'reason' in signal:
print(f" Reason: {signal['reason']}")
Part 2: Social Media Sentiment
Twitter/X Sentiment Analysis
import tweepy
from textblob import TextBlob
import pandas as pd
from collections import Counter
from datetime import datetime, timedelta
class TwitterSentimentTracker:
"""
Track stock sentiment on Twitter/X
Limitations:
- Twitter API v2 requires paid access for historical data
- Free tier: 10,000 tweets/month
- Use for stocks with high social media buzz
"""
def __init__(self, bearer_token: str):
self.client = tweepy.Client(bearer_token=bearer_token)
def search_tweets(self, query: str, max_results: int = 100) -> List[Dict]:
"""
Search recent tweets (last 7 days with free tier)
Query examples:
- "$RELIANCE" (cashtag)
- "Reliance Industries"
- "Nifty 50"
"""
try:
# Search recent tweets
tweets = self.client.search_recent_tweets(
query=query,
max_results=max_results,
tweet_fields=['created_at', 'public_metrics', 'lang']
)
if not tweets.data:
return []
tweet_list = []
for tweet in tweets.data:
tweet_list.append({
'id': tweet.id,
'text': tweet.text,
'created_at': tweet.created_at,
'likes': tweet.public_metrics['like_count'],
'retweets': tweet.public_metrics['retweet_count'],
'lang': tweet.lang
})
return tweet_list
except Exception as e:
print(f"Error fetching tweets: {e}")
return []
def analyze_sentiment(self, tweets: List[Dict]) -> Dict:
"""
Analyze sentiment using TextBlob
Polarity: -1 (negative) to +1 (positive)
"""
sentiments = []
for tweet in tweets:
# Skip non-English tweets
if tweet['lang'] != 'en':
continue
# Analyze sentiment
blob = TextBlob(tweet['text'])
polarity = blob.sentiment.polarity
# Weight by engagement
engagement = tweet['likes'] + (tweet['retweets'] * 2)
weight = 1 + np.log1p(engagement)
sentiments.append({
'text': tweet['text'],
'polarity': polarity,
'weight': weight,
'weighted_polarity': polarity * weight
})
if not sentiments:
return {'sentiment_score': 0, 'signal': 'NO_DATA'}
# Calculate weighted average sentiment
total_weight = sum(s['weight'] for s in sentiments)
weighted_sentiment = sum(s['weighted_polarity'] for s in sentiments) / total_weight
# Classify sentiment
if weighted_sentiment > 0.2:
signal = 'BULLISH'
elif weighted_sentiment < -0.2:
signal = 'BEARISH'
else:
signal = 'NEUTRAL'
# Distribution
positive = sum(1 for s in sentiments if s['polarity'] > 0)
negative = sum(1 for s in sentiments if s['polarity'] < 0)
neutral = len(sentiments) - positive - negative
return {
'total_tweets': len(sentiments),
'positive_tweets': positive,
'negative_tweets': negative,
'neutral_tweets': neutral,
'sentiment_score': weighted_sentiment,
'signal': signal,
'confidence': abs(weighted_sentiment)
}
def generate_trading_signal(self, symbol: str) -> Dict:
"""
Generate trading signal from Twitter sentiment
"""
# Search for stock mentions
query = f"${symbol} OR \"{symbol}\" lang:en -is:retweet"
tweets = self.search_tweets(query, max_results=100)
if not tweets:
return {'signal': 'NO_DATA', 'reason': 'No tweets found'}
# Analyze sentiment
sentiment = self.analyze_sentiment(tweets)
return {
'symbol': symbol,
'timestamp': datetime.now(),
'tweets_analyzed': sentiment['total_tweets'],
'sentiment_score': sentiment['sentiment_score'],
'signal': sentiment['signal'],
'confidence': sentiment['confidence'],
'distribution': {
'positive': sentiment['positive_tweets'],
'negative': sentiment['negative_tweets'],
'neutral': sentiment['neutral_tweets']
}
}
# Example (requires Twitter API credentials)
# tracker = TwitterSentimentTracker(bearer_token='your_token')
# signal = tracker.generate_trading_signal('RELIANCE')
#
# print(f"\n🐦 Twitter Sentiment: {signal['symbol']}")
# print(f" Signal: {signal['signal']}")
# print(f" Sentiment Score: {signal['sentiment_score']:.2f}")
# print(f" Tweets Analyzed: {signal['tweets_analyzed']}")
Part 3: Google Trends Analysis
Search Interest as Leading Indicator
from pytrends.request import TrendReq
import pandas as pd
import numpy as np
class GoogleTrendsAnalyzer:
"""
Analyze Google search trends for trading signals
Theory: Increased search interest precedes price movements
"""
def __init__(self):
self.pytrends = TrendReq(hl='en-IN', tz=330)
def get_interest_over_time(self, keyword: str, timeframe: str = 'today 3-m') -> pd.DataFrame:
"""
Get search interest time series
Timeframe options:
- 'today 1-m': Last 30 days
- 'today 3-m': Last 90 days
- 'today 12-m': Last year
"""
try:
# Build payload
self.pytrends.build_payload([keyword], timeframe=timeframe, geo='IN')
# Get data
data = self.pytrends.interest_over_time()
if data.empty:
return pd.DataFrame()
# Remove 'isPartial' column
if 'isPartial' in data.columns:
data = data.drop(columns=['isPartial'])
return data
except Exception as e:
print(f"Error fetching trends: {e}")
return pd.DataFrame()
def calculate_trend_momentum(self, data: pd.DataFrame, keyword: str) -> Dict:
"""
Calculate momentum indicators from search trends
"""
if data.empty or keyword not in data.columns:
return {'momentum': 0, 'signal': 'NO_DATA'}
values = data[keyword].values
# Calculate metrics
current = values[-1]
week_ago = values[-7] if len(values) >= 7 else values[0]
month_ago = values[-30] if len(values) >= 30 else values[0]
# Short-term momentum (1 week)
weekly_change = (current - week_ago) / week_ago if week_ago > 0 else 0
# Long-term momentum (1 month)
monthly_change = (current - month_ago) / month_ago if month_ago > 0 else 0
# Moving average comparison
ma_short = values[-7:].mean() if len(values) >= 7 else current
ma_long = values[-30:].mean() if len(values) >= 30 else current
# Combined momentum score
momentum_score = (weekly_change * 0.4) + (monthly_change * 0.3) + \
((ma_short - ma_long) / ma_long * 0.3 if ma_long > 0 else 0)
# Generate signal
if momentum_score > 0.2:
signal = 'INCREASING_INTEREST' # Bullish
elif momentum_score < -0.2:
signal = 'DECREASING_INTEREST' # Bearish
else:
signal = 'STABLE'
return {
'current_interest': current,
'weekly_change': weekly_change,
'monthly_change': monthly_change,
'momentum_score': momentum_score,
'signal': signal
}
def compare_stocks(self, keywords: List[str], timeframe: str = 'today 3-m') -> pd.DataFrame:
"""
Compare search interest across multiple stocks
"""
try:
self.pytrends.build_payload(keywords, timeframe=timeframe, geo='IN')
data = self.pytrends.interest_over_time()
if 'isPartial' in data.columns:
data = data.drop(columns=['isPartial'])
return data
except Exception as e:
print(f"Error comparing trends: {e}")
return pd.DataFrame()
# Example usage
analyzer = GoogleTrendsAnalyzer()
# Analyze Reliance search interest
keyword = "Reliance Industries share price"
data = analyzer.get_interest_over_time(keyword, timeframe='today 3-m')
if not data.empty:
momentum = analyzer.calculate_trend_momentum(data, keyword)
print(f"\n🔍 Google Trends: {keyword}")
print(f" Current Interest: {momentum['current_interest']}")
print(f" Weekly Change: {momentum['weekly_change']:.2%}")
print(f" Monthly Change: {momentum['monthly_change']:.2%}")
print(f" Signal: {momentum['signal']}")
# Plot
data[keyword].plot(figsize=(12, 5), title='Search Interest Over Time')
plt.ylabel('Interest')
plt.grid(True, alpha=0.3)
plt.show()
Part 4: Combining Alternative Data
Multi-Source Signal Aggregation
class AlternativeDataAggregator:
"""
Combine signals from multiple alternative data sources
"""
def __init__(self):
self.weights = {
'announcements': 0.3,
'twitter_sentiment': 0.3,
'google_trends': 0.2,
'technical': 0.2
}
def aggregate_signals(self, signals: Dict[str, Dict]) -> Dict:
"""
Weighted aggregation of alternative data signals
Args:
signals: Dict with keys matching self.weights
"""
# Normalize signals to [-1, 1] scale
normalized_signals = {}
# Announcements
if 'announcements' in signals:
ann_signal = signals['announcements']
if ann_signal['signal'] == 'BUY':
normalized_signals['announcements'] = 1.0
elif ann_signal['signal'] == 'SELL':
normalized_signals['announcements'] = -1.0
else:
normalized_signals['announcements'] = 0.0
# Twitter sentiment
if 'twitter_sentiment' in signals:
tw_signal = signals['twitter_sentiment']
# Already normalized (-1 to 1)
normalized_signals['twitter_sentiment'] = tw_signal.get('sentiment_score', 0)
# Google trends
if 'google_trends' in signals:
gt_signal = signals['google_trends']
momentum = gt_signal.get('momentum_score', 0)
# Clip to [-1, 1]
normalized_signals['google_trends'] = np.clip(momentum, -1, 1)
# Calculate weighted score
total_score = 0
total_weight = 0
for source, score in normalized_signals.items():
weight = self.weights.get(source, 0)
total_score += score * weight
total_weight += weight
final_score = total_score / total_weight if total_weight > 0 else 0
# Generate final signal
if final_score > 0.3:
final_signal = 'BUY'
elif final_score < -0.3:
final_signal = 'SELL'
else:
final_signal = 'NEUTRAL'
return {
'final_signal': final_signal,
'score': final_score,
'confidence': abs(final_score),
'contributing_sources': len(normalized_signals),
'source_scores': normalized_signals
}
# Example: Combine all signals for Reliance
aggregator = AlternativeDataAggregator()
combined_signals = {
'announcements': {'signal': 'BUY', 'reason': 'Dividend announcement'},
'twitter_sentiment': {'sentiment_score': 0.4},
'google_trends': {'momentum_score': 0.25}
}
result = aggregator.aggregate_signals(combined_signals)
print(f"\n📊 Alternative Data Consensus:")
print(f" Final Signal: {result['final_signal']}")
print(f" Score: {result['score']:.2f}")
print(f" Confidence: {result['confidence']:.2f}")
print(f" Sources: {result['contributing_sources']}")
Conclusion
Alternative data provides an edge—but only when used correctly:
- Verify legality - ensure compliance with Indian regulations
- Cross-validate - never rely on a single source
- Backt test - validate that alternative data actually predicts prices
- Monitor staleness - data advantages decay as others discover them
- Cost-benefit - paid data must generate alpha exceeding costs
Start with free sources (web scraping, Google Trends, public social media) before investing in expensive satellite or credit card data.
Ready to build an alternative data strategy? Contact us for custom data sourcing and signal extraction.