सांख्यिकीय सारांश एजेंट
एजेंट द्वारा बनाए गए रिपोर्ट में वितरण विश्लेषण, सहसंबंध और बाहरी मानों की पहचान।
सांख्यिकीय सारांश एजेंट, CoddyKit पर AI एजेंट का एक निःशुल्क पाठ है। यह 4 में से 4वाँ पाठ है। आप नीचे पूरा पाठ निःशुल्क पढ़ सकते हैं—फिर अंतर्निहित कोड संपादक और 24/7 एआई ट्यूटर के साथ ब्राउज़र में इसका व्यावहारिक अभ्यास कर सकते हैं। यह AI एजेंट सीखने के मार्ग का हिस्सा है और आपकी प्रगति वेब तथा CoddyKit ऐप पर सिंक होती रहती है। AI एजेंट पाठ्यक्रम में कुल 4 पाठ शामिल हैं।
सांख्यिकीय सारांश क्यों महत्वपूर्ण हैं
कच्चा डेटा अपने-आप में शायद ही उपयोगी होता है। डेटा विश्लेषण एजेंट तब वास्तव में उपयोगी बनता है जब वह सांख्यिकीय सारांश अपने-आप निकाल सके और संख्याओं को सरल भाषा के निष्कर्षों में बदल सके।
इस पाठ में मुख्य सांख्यिकीय कार्य शामिल हैं: describe, सहसंबंध, बाहरी मानों की पहचान, वितरण की पहचान और स्वचालित निष्कर्ष निर्माण।
df.describe(): आरंभिक बिंदु
df.describe() एक ही कॉल में सभी संख्यात्मक कॉलमों के लिए गणना, mean, std, min, चतुर्थक और max निकालता है। यह किसी भी अन्वेषणात्मक डेटा विश्लेषण का सामान्य पहला चरण है।
import pandas as pd
df = pd.read_csv('sales_data.csv')
# Basic numeric summary
print(df.describe())
# Also describe categorical columns
print(df.describe(include='object'))
# Custom percentiles
print(df.describe(percentiles=[0.05, 0.25, 0.5, 0.75, 0.95]))
# Convert to clean dict for agent use
def get_numeric_summary(df):
desc = df.describe().round(2)
return {
col: desc[col].to_dict()
for col in desc.columns
}सहसंबंध मैट्रिक्स
सहसंबंध मैट्रिक्स सभी संख्यात्मक कॉलमों के बीच युग्म-दर-युग्म रैखिक संबंध दिखाता है। मान -1 (मज़बूत नकारात्मक संबंध) से +1 (मज़बूत सकारात्मक संबंध) तक होते हैं। 0 के निकट मान किसी रैखिक संबंध की अनुपस्थिति दर्शाते हैं।
import pandas as pd
import numpy as np
def compute_correlation_matrix(df):
numeric = df.select_dtypes(include='number')
corr = numeric.corr().round(3)
return corr.to_dict()
def find_strong_correlations(df, threshold=0.7):
numeric = df.select_dtypes(include='number')
corr = numeric.corr()
strong = []
cols = corr.columns
for i in range(len(cols)):
for j in range(i + 1, len(cols)): # upper triangle only
val = corr.iloc[i, j]
if abs(val) >= threshold:
strong.append({
'col_a': cols[i],
'col_b': cols[j],
'correlation': round(val, 3),
'direction': 'positive' if val > 0 else 'negative'
})
return sorted(strong, key=lambda x: abs(x['correlation']), reverse=True)बाहरी मानों की पहचान: IQR विधि
अंतरचतुर्थक परास (IQR) विधि Q3 से 1.5x IQR से अधिक ऊपर या Q1 से नीचे के मानों को बाहरी मान के रूप में पहचानती है। यह अत्यधिक मानों के प्रति सुदृढ़ होती है—z-स्कोर के विपरीत, डेटा में बाहरी मान मौजूद होने पर भी यह अच्छी तरह काम करती है।
import pandas as pd
def detect_outliers_iqr(df, columns=None, multiplier=1.5):
numeric = df.select_dtypes(include='number')
if columns:
numeric = numeric[columns]
outlier_report = {}
for col in numeric.columns:
q1 = numeric[col].quantile(0.25)
q3 = numeric[col].quantile(0.75)
iqr = q3 - q1
lower = q1 - multiplier * iqr
upper = q3 + multiplier * iqr
outliers = numeric[(numeric[col] < lower) | (numeric[col] > upper)][col]
outlier_report[col] = {
'q1': round(q1, 3),
'q3': round(q3, 3),
'iqr': round(iqr, 3),
'lower_bound': round(lower, 3),
'upper_bound': round(upper, 3),
'outlier_count': len(outliers),
'outlier_pct': round(len(outliers) / len(df) * 100, 1),
'outlier_values': outliers.tolist()[:10] # first 10
}
return outlier_reportवितरण प्रकार की पहचान
वितरण प्रकार (सामान्य, विषम-वितरित, द्विशिखरी, समरूप) की पहचान करने से एजेंट सही सांख्यिकीय परीक्षण चुन सकता है और डेटा का सटीक वर्णन कर सकता है।
औपचारिक परीक्षण चलाने से पहले तेज़ अनुमान के रूप में विषमता और kurtosis का उपयोग करें।
import pandas as pd
from scipy import stats
def identify_distribution(series):
'''Classify distribution type using skewness, kurtosis, and normality test.'''
clean = series.dropna()
if len(clean) < 8:
return {'type': 'insufficient_data'}
skewness = clean.skew()
kurtosis = clean.kurtosis()
# Shapiro-Wilk normality test (reliable up to ~5000 samples)
sample = clean.sample(min(500, len(clean)), random_state=42)
_, p_value = stats.shapiro(sample)
distribution = 'normal' if p_value > 0.05 else 'non-normal'
if abs(skewness) < 0.5 and p_value > 0.05:
dist_type = 'normal'
elif skewness > 1.0:
dist_type = 'right-skewed (long right tail)'
elif skewness < -1.0:
dist_type = 'left-skewed (long left tail)'
elif abs(skewness) < 0.5 and kurtosis < -1:
dist_type = 'uniform-like'
else:
dist_type = 'approximately normal'
return {
'type': dist_type,
'skewness': round(skewness, 3),
'kurtosis': round(kurtosis, 3),
'normality_pvalue': round(p_value, 4),
'is_normal': p_value > 0.05
}सामान्यता परीक्षण
कई औपचारिक सामान्यता परीक्षण उपलब्ध हैं। Shapiro-Wilk छोटे नमूनों के लिए सबसे अच्छा है; D'Agostino-Pearson बड़े डेटासेट के लिए बेहतर है। p-मान की सीमा आम तौर पर 0.05 होती है—यदि p < 0.05 हो, तो सामान्यता को अस्वीकार करें।
from scipy import stats
import numpy as np
def test_normality(series, alpha=0.05):
clean = series.dropna().values
n = len(clean)
results = {}
# Shapiro-Wilk (best for n < 5000)
if n <= 5000:
stat, p = stats.shapiro(clean)
results['shapiro_wilk'] = {
'statistic': round(stat, 4),
'p_value': round(p, 4),
'is_normal': p > alpha
}
# D'Agostino-Pearson (n >= 20)
if n >= 20:
stat, p = stats.normaltest(clean)
results['dagostino_pearson'] = {
'statistic': round(stat, 4),
'p_value': round(p, 4),
'is_normal': p > alpha
}
# Kolmogorov-Smirnov
mean, std = np.mean(clean), np.std(clean)
stat, p = stats.kstest(clean, 'norm', args=(mean, std))
results['kolmogorov_smirnov'] = {
'statistic': round(stat, 4),
'p_value': round(p, 4),
'is_normal': p > alpha
}
return resultsपूर्ण सांख्यिकीय प्रोफ़ाइल
सभी सांख्यिकीय विश्लेषणों को एक ही full_profile() फ़ंक्शन में मिलाएँ, जो डेटाफ़्रेम के प्रत्येक संख्यात्मक कॉलम के लिए एक व्यापक सांख्यिकीय सारांश लौटाता है।
def full_statistical_profile(df):
numeric = df.select_dtypes(include='number')
profiles = {}
for col in numeric.columns:
series = numeric[col].dropna()
profile = {
'count': len(series),
'null_count': df[col].isnull().sum(),
'mean': round(series.mean(), 4),
'median': round(series.median(), 4),
'std': round(series.std(), 4),
'min': round(series.min(), 4),
'max': round(series.max(), 4),
'range': round(series.max() - series.min(), 4),
'q1': round(series.quantile(0.25), 4),
'q3': round(series.quantile(0.75), 4)
}
profile['distribution'] = identify_distribution(series)
outliers = detect_outliers_iqr(df, columns=[col])
profile['outliers'] = outliers.get(col, {})
profiles[col] = profile
return {
'profiles': profiles,
'correlations': find_strong_correlations(df),
'shape': {'rows': len(df), 'columns': len(df.columns)}
}स्वचालित निष्कर्ष निर्माण
सांख्यिकीय प्रोफ़ाइल LLM को दें और उससे सरल भाषा में निष्कर्ष बनाने को कहें। मॉडल पैटर्न पहचानता है, असामान्यताओं को चिह्नित करता है और वे निष्कर्ष सामने लाता है जिनमें व्यावसायिक उपयोगकर्ता की रुचि होगी।
import json
INSIGHT_PROMPT = '''You are a data analyst. Generate 3-5 key insights from this statistical summary.
Dataset: {dataset_name}
Statistical profile:
{profile_json}
For each insight:
- Be specific (use actual numbers)
- Focus on business relevance
- Flag any data quality concerns (high null counts, extreme outliers)
- Note unexpected patterns
Format as bullet points.'''
def generate_insights(df, dataset_name='dataset'):
profile = full_statistical_profile(df)
profile_text = json.dumps(profile, indent=2, default=str)[:3000]
return llm_call(INSIGHT_PROMPT.format(
dataset_name=dataset_name,
profile_json=profile_text
))
# Example output:
# - Revenue column has 3.2% outliers; max value ($48,500) is 12x the median ($3,900)
# - Customer age follows a right-skewed distribution (skewness: 1.4) with most users under 35
# - Strong positive correlation (r=0.87) between session_count and lifetime_valueसमय-श्रृंखला प्रवृत्ति की पहचान
समय-श्रृंखला डेटा के लिए रैखिक प्रतिगमन का उपयोग करके समग्र प्रवृत्ति (बढ़ती, घटती, स्थिर) की पहचान करें। इससे एजेंट यह कह सकते हैं: "राजस्व हर महीने $2,400 बढ़ रहा है।"
import numpy as np
from scipy import stats
def detect_trend(values):
if len(values) < 3:
return {'trend': 'insufficient_data'}
x = np.arange(len(values))
y = np.array(values, dtype=float)
slope, intercept, r_value, p_value, std_err = stats.linregress(x, y)
if p_value > 0.05:
trend = 'flat (no significant trend)'
elif slope > 0:
trend = 'increasing'
else:
trend = 'decreasing'
return {
'trend': trend,
'slope_per_period': round(slope, 4),
'r_squared': round(r_value ** 2, 4),
'p_value': round(p_value, 4),
'significant': p_value < 0.05
}
# Example: monthly revenue trend
monthly_revenue = [10000, 11200, 10800, 12500, 13100, 14200, 15000]
print(detect_trend(monthly_revenue))
# {'trend': 'increasing', 'slope_per_period': 834.2, 'r_squared': 0.95, ...}तुलनात्मक सांख्यिकी
अक्सर सबसे उपयोगी निष्कर्ष समूहों की तुलना से मिलता है: "सेगमेंट A का औसत ऑर्डर मूल्य सेगमेंट B से 40% अधिक है।" समूहों की तुलना के लिए t-परीक्षण या Mann-Whitney का उपयोग करें।
from scipy import stats
import pandas as pd
def compare_groups(df, group_column, value_column):
groups = df.groupby(group_column)[value_column].apply(list)
group_names = list(groups.index)
comparison = []
for i, g1 in enumerate(group_names):
for j, g2 in enumerate(group_names):
if j <= i:
continue
a, b = groups[g1], groups[g2]
mean_a = pd.Series(a).mean()
mean_b = pd.Series(b).mean()
# Mann-Whitney U (non-parametric, no normality assumption)
stat, p = stats.mannwhitneyu(a, b, alternative='two-sided')
pct_diff = (mean_a - mean_b) / mean_b * 100 if mean_b != 0 else 0
comparison.append({
'group_a': g1, 'group_b': g2,
'mean_a': round(mean_a, 2),
'mean_b': round(mean_b, 2),
'pct_difference': round(pct_diff, 1),
'p_value': round(p, 4),
'significant': p < 0.05
})
return comparisonअनुपलब्ध मानों का विश्लेषण
कोई भी सांख्यिकी निकालने से पहले अनुपलब्ध मानों का विश्लेषण करें। किसी कॉलम में रिक्त मानों की अधिक दर उससे निकाली गई सांख्यिकी को अमान्य कर देती है। रिक्त मानों की गिनती, प्रतिशत और पैटर्न की रिपोर्ट करें (क्या रिक्त मान यादृच्छिक हैं या किसी व्यवस्थित कारण से?)।
import pandas as pd
import numpy as np
def analyze_missing_values(df):
total_rows = len(df)
missing_report = {}
for col in df.columns:
null_count = df[col].isnull().sum()
if null_count == 0:
continue
null_pct = null_count / total_rows * 100
# Check if nulls correlate with another column (systematic pattern)
pattern = 'random'
if null_pct > 5:
for other_col in df.select_dtypes(include='object').columns:
if other_col == col:
continue
null_rates = df.groupby(other_col)[col].apply(
lambda x: x.isnull().mean()
)
if null_rates.max() - null_rates.min() > 0.3: # 30% difference
pattern = f'correlated_with_{other_col}'
break
missing_report[col] = {
'null_count': int(null_count),
'null_pct': round(null_pct, 1),
'severity': 'high' if null_pct > 20 else 'medium' if null_pct > 5 else 'low',
'pattern': pattern
}
return missing_reportज्ञान-जाँच
एजेंट अनुप्रयोगों में बाहरी मानों की पहचान के लिए z-स्कोर की तुलना में IQR विधि को प्राथमिकता क्यों दी जाती है?
पुनरावलोकन: सांख्यिकीय सारांश एजेंट
सांख्यिकीय सारांश एजेंट इनका संयोजन करते हैं: आधारभूत आँकड़ों के लिए df.describe(), संबंधों की खोज के लिए सहसंबंध मैट्रिक्स, IQR से बाहरी मानों की पहचान (अत्यधिक मानों के प्रति सुदृढ़), विषमता और सामान्यता परीक्षणों के माध्यम से वितरण की पहचान, तथा रैखिक प्रतिगमन के माध्यम से प्रवृत्ति की पहचान।
स्वचालित निष्कर्ष निर्माण के लिए सांख्यिकीय प्रोफ़ाइल LLM को दें—इससे संख्याओं को सरल भाषा में व्यावसायिक निष्कर्षों में बदला जा सकेगा। तुलनात्मक सांख्यिकी (महत्त्व-परीक्षण के साथ समूह तुलना) सबसे उपयोगी निष्कर्ष सामने लाती है।
एआई शिक्षक के साथ AI एजेंट सीखें — निःशुल्क
अपने ब्राउज़र में वास्तविक कोड लिखें और चलाएँ, चौबीसों घंटे एआई शिक्षक से तुरंत सहायता पाएँ, और वेब या ऐप पर वहीं से शुरू करें जहाँ आपने छोड़ा था।
- पाठ्यक्रम
- 60
- पाठ
- 239
अक्सर पूछे जाने वाले प्रश्न
क्या “सांख्यिकीय सारांश एजेंट” पाठ निःशुल्क है?
हाँ—“सांख्यिकीय सारांश एजेंट” का पूरा पाठ यहाँ वेब पर निःशुल्क पढ़ा जा सकता है। इंटरैक्टिव अभ्यास (अंतर्निहित कोड संपादक और 24/7 एआई ट्यूटर) करने और AI एजेंट पाठ्यक्रम का बाकी हिस्सा अनलॉक करने के लिए CoddyKit PRO लें। AI एजेंट पाठ्यक्रम में कुल 4 पाठ शामिल हैं।
“सांख्यिकीय सारांश एजेंट” में मैं क्या सीखूँगा?
एजेंट द्वारा बनाए गए रिपोर्ट में वितरण विश्लेषण, सहसंबंध और बाहरी मानों की पहचान। आप ब्राउज़र में सीधे चलाए जाने वाले व्यावहारिक कोड के साथ AI एजेंट का अभ्यास करते हैं, और पाठ पूरा करते समय 24/7 एआई ट्यूटर आपके प्रश्नों के उत्तर देता है।
क्या AI एजेंट शुरू करने के लिए मुझे किसी अनुभव की आवश्यकता है?
पहले के अनुभव की आवश्यकता नहीं है। CoddyKit पर AI एजेंट शुरुआती से लेकर उन्नत शिक्षार्थियों तक सभी के लिए व्यवस्थित किया गया है, इसलिए आप यहीं से या शुरुआत से सीखना शुरू कर सकते हैं और अपनी गति से आगे बढ़ सकते हैं। यह 4 में से 4वाँ पाठ है।
“सांख्यिकीय सारांश एजेंट” पाठ पूरा करने में कितना समय लगता है?
CoddyKit का अधिकांश पाठ लगभग 5–10 मिनट में पूरा हो जाता है। हर पाठ छोटा और संवादात्मक है, इसलिए आप लगातार प्रगति करते हैं और वेब या ऐप पर वहीं से सीखना जारी रख सकते हैं जहाँ आपने छोड़ा था।
क्या मैं इस AI एजेंट पाठ में कोड लिख और चला सकता हूँ?
हाँ। हर AI एजेंट पाठ में एक अंतर्निर्मित कोड संपादक शामिल है, जिससे आप सीधे अपने ब्राउज़र में वास्तविक कोड लिख और चला सकते हैं और तुरंत एआई प्रतिक्रिया पा सकते हैं—स्थानीय सेटअप की आवश्यकता नहीं है।
इस पाठ्यक्रम के सभी पाठ
- डेटा विश्लेषण के लिए कोड इंटरप्रेटर पैटर्न
- Pandas-आधारित डेटा एजेंट टूल
- स्वचालित चार्ट और विज़ुअलाइज़ेशन बनाना
- सांख्यिकीय सारांश एजेंट