Algorithms in ySights
This tutorial covers the analytical algorithms available in ySights for understanding simulation dynamics.
What You’ll Learn
Profile similarity analysis
Semantic text enrichment and similarity checks
Sentiment diffusion and recommendation exposure
Topic lifecycle analysis
Multiplex interaction diagnostics
Moderation and forum session summaries
[ ]:
from pathlib import Path
from ysights import YDataHandler
from ysights.algorithms import (
profile_topics_similarity,
visibility_paradox,
user_visibility_vs_neighbors,
visibility_paradox_population_size_null,
engagement_momentum,
personalization_balance_score,
sentiment_diffusion_metrics,
)
from ysights.algorithms.topics import topic_spread, adoption_rate, peak_engagement_time
import matplotlib.pyplot as plt
import numpy as np
plt.style.use('seaborn-v0_8-darkgrid')
%matplotlib inline
[ ]:
# Initialize data handler and get network
from pathlib import Path
def resolve_example_db():
candidates = [
Path("ysocial_db.db"),
Path("../notebooks/ysocial_db.db"),
Path("../../notebooks/ysocial_db.db"),
Path("docs/notebooks/ysocial_db.db"),
]
for candidate in candidates:
if candidate.exists():
return str(candidate.resolve())
return "ysocial_db.db"
db_path = resolve_example_db()
ydh = YDataHandler(db_path)
network = ydh.social_network()
1. Profile Similarity Analysis
Analyzes how similar users’ interest profiles are across the network.
[ ]:
similarity_scores = profile_topics_similarity(ydh, network)
similarity_values = list(similarity_scores.values())
print(f"Computed {len(similarity_scores)} similarity scores")
print(f"\nSimilarity Statistics:")
print(f" Mean: {np.mean(similarity_values):.4f}")
print(f" Median: {np.median(similarity_values):.4f}")
print(f" Std Dev: {np.std(similarity_values):.4f}")
print(f" Min: {min(similarity_values):.4f}")
print(f" Max: {max(similarity_values):.4f}")
[ ]:
plt.figure(figsize=(12, 5))
plt.subplot(1, 2, 1)
plt.hist(similarity_values, bins=50, edgecolor='black', alpha=0.7, color='coral')
plt.xlabel('Similarity Score', fontsize=11)
plt.ylabel('Frequency', fontsize=11)
plt.title('Profile Similarity Distribution', fontsize=13, fontweight='bold')
plt.axvline(np.mean(similarity_values), color='red', linestyle='--', label=f'Mean: {np.mean(similarity_values):.3f}')
plt.legend()
plt.grid(True, alpha=0.3)
plt.subplot(1, 2, 2)
plt.boxplot(similarity_values, vert=True)
plt.ylabel('Similarity Score', fontsize=11)
plt.title('Profile Similarity Box Plot', fontsize=13, fontweight='bold')
plt.grid(True, alpha=0.3, axis='y')
plt.tight_layout()
plt.show()
2. Recommendation System Metrics
Engagement Momentum
Measures how consistently users engage with recommended content over time.
[ ]:
momentum = engagement_momentum(ydh, time_window_rounds=24)
print("Engagement Momentum Analysis:")
print(f" Users analyzed: {len(momentum)}")
print(f" Average momentum: {np.mean(list(momentum.values())):.4f}")
print(f" Median momentum: {np.median(list(momentum.values())):.4f}")
top_momentum = sorted(momentum.items(), key=lambda x: x[1], reverse=True)[:5]
print("\nTop 5 Users by Engagement Momentum:")
for i, (user, score) in enumerate(top_momentum, 1):
print(f" {i}. User {user}: {score:.4f}")
Personalization Balance Score
Measures how well the recommendation system balances exploration vs. exploitation.
[ ]:
balance_scores = personalization_balance_score(ydh)
print("Personalization Balance Analysis:")
print(f" Users analyzed: {len(balance_scores)}")
print(f" Average balance: {np.mean(list(balance_scores.values())):.4f}")
print(f" Median balance: {np.median(list(balance_scores.values())):.4f}")
print("\nInterpretation:")
print(" Score → 0: Heavy exploitation (narrow recommendations)")
print(" Score → 1: Heavy exploration (diverse recommendations)")
[ ]:
plt.figure(figsize=(10, 6))
plt.hist(list(balance_scores.values()), bins=30, edgecolor='black', alpha=0.7, color='seagreen')
plt.xlabel('Personalization Balance Score', fontsize=12)
plt.ylabel('Number of Users', fontsize=12)
plt.title('Distribution of Personalization Balance', fontsize=14, fontweight='bold')
plt.axvline(0.5, color='red', linestyle='--', linewidth=2, label='Perfect Balance')
plt.legend()
plt.grid(True, alpha=0.3)
plt.tight_layout()
plt.show()
3. Topic Lifecycle Analysis
These helpers are thin wrappers around the implemented topic lifecycle analysis in YDataHandler.
[ ]:
lifecycles = topic_spread(ydh)
adoption_rates = adoption_rate(ydh)
peak_periods = peak_engagement_time(ydh)
topic_ids = list(lifecycles.keys())
print(f"Topics analyzed: {len(topic_ids)}")
for topic_id in topic_ids[:5]:
lifecycle = lifecycles[topic_id]
print(
f" Topic {topic_id}: posts={lifecycle['post_count']}, "
f"authors={lifecycle['author_count']}, "
f"peak={lifecycle['peak_period']}, "
f"adoption={adoption_rates[topic_id]:.3f}"
)
if topic_ids:
topic_id = topic_ids[0]
print(f"\nTimeline preview for topic {topic_id}:")
print(lifecycles[topic_id]["timeline"].head().to_string(index=False))
4. Semantic and Content Diagnostics
These helpers expose lightweight semantic features, text normalization, and similarity checks that work across content sources.
Text normalization and semantic profiles
Normalize a post, inspect its semantic profile, and compare it with a short sample string.
[ ]:
agent_id = next(iter(ydh.agent_mapping()))
sample_posts = ydh.posts_by_agent(agent_id).get_posts()
if sample_posts:
sample_post = sample_posts[0]
post_profile = ydh.post_semantic_profile(sample_post.id)
print(f"Semantic profile for post {sample_post.id}:")
for key in [
"token_count",
"word_count",
"lexical_diversity",
"readability_proxy",
"punctuation_intensity",
]:
print(f" {key}: {post_profile[key]}")
similarity = ydh.semantic_similarity(
sample_post.text,
"Short reply about a similar topic",
)
print("\nSemantic similarity against a short sample:")
print(f" mode: {similarity['mode']}")
print(f" score: {similarity['score']:.4f}")
print(f" token_jaccard: {similarity['token_jaccard']:.4f}")
else:
print("No sample posts were found in this dataset.")
Exercise
Compare two short content snippets and inspect how lexical similarity changes when the wording changes only slightly.
[ ]:
exercise_similarity = ydh.semantic_similarity(
"Forum users share updates",
"Users share updates in the forum",
)
print("Exercise result:")
print(f" mode: {exercise_similarity['mode']}")
print(f" score: {exercise_similarity['score']:.4f}")
print(f" token_jaccard: {exercise_similarity['token_jaccard']:.4f}")
Sentiment diffusion and recommendation exposure
Summarize how sentiment propagates through recommendations and how often the same exposure pairs recur.
[ ]:
sentiment = sentiment_diffusion_metrics(ydh)
exposure = ydh.recommendation_exposure_summary()
print("Sentiment diffusion:")
for key in ["available", "post_count", "labeled_post_count", "sentiment_coverage", "recommendation_count"]:
if key in sentiment:
print(f" {key}: {sentiment[key]}")
print(" sentiment_label_counts:", sentiment["sentiment_label_counts"])
print(" average_scores:", sentiment["average_scores"])
print("\nSentiment diffusion timeline preview:")
print(sentiment["timeline"].head().to_string(index=False))
print("\nRecommendation exposure summary:")
for key in ["exposure_count", "unique_recipients", "unique_posts", "unique_authors"]:
print(f" {key}: {exposure[key]}")
print(" conversion_rates:", exposure["conversion_rates"])
print(" feedback_loop:", exposure["feedback_loop"])
print("\nExposure timeline preview:")
print(exposure["timeline"].head().to_string(index=False))
Multiplex interaction overview
Inspect the available interaction layers and the dominant nodes in each layer.
[ ]:
multiplex = ydh.multiplex_metrics()
print(f"Available layers: {multiplex['available_layers']}")
for layer_name, layer_metrics in multiplex["layer_metrics"].items():
print(f" {layer_name}: {layer_metrics['edge_count']} edges, weight_sum={layer_metrics['weight_sum']}")
centrality = layer_metrics["centrality"]
print(f" top in-degree node: {centrality['top_in_degree']['node']}")
print(f" strong tie share: {layer_metrics['tie_strength']['strong_tie_share']:.4f}")
print("\nPairwise overlap preview:")
for pair_name, overlap in list(multiplex["pairwise_overlap"].items())[:3]:
print(
f" {pair_name}: edge_jaccard={overlap['edge_jaccard']:.4f}, "
f"node_jaccard={overlap['node_jaccard']:.4f}"
)
print("\nCombined polarization:")
print(multiplex["combined_polarization"])
5. Operational and Moderation Summaries
These helpers are useful when you want a compact dataset overview before running a deeper analysis.
[ ]:
summary_report = ydh.summary_report()
summary_frame = ydh.summary_frame()
moderation_summary = ydh.moderation_summary()
forum_sessions = ydh.forum_session_summaries()
interaction_layers = ydh.interaction_layers()
multiplex = ydh.multiplex_metrics()
cache_info = ydh.analysis_cache_info()
recommended_indexes = ydh.recommended_indexes()
benchmark = ydh.benchmark_analytics(iterations=1)
print("Summary Report (selected keys):")
for key in ["post_count", "thread_count", "report_count", "forum_session_count", "moderated_posts"]:
if key in summary_report:
print(f" {key}: {summary_report[key]}")
print("\nInteraction Layer Summary:")
print(f" layers: {list(interaction_layers.keys())}")
print(f" combined edges: {multiplex['combined']['edge_count']}")
print(f" reciprocity: {multiplex['combined_polarization']['reciprocity']}")
print("\nSummary Frame Preview:")
print(summary_frame.head().to_string(index=False))
print("\nModeration Summary:")
print(moderation_summary)
print("\nForum Session Summaries:")
print(forum_sessions)
print("\nCache Diagnostics:")
print(cache_info)
print("\nRecommended Indexes:")
print(recommended_indexes)
print("\nBenchmark Targets:")
print(list(benchmark["metrics"].keys()))
Summary
In this tutorial, you learned:
✓ How to measure profile similarity across the network ✓ How to normalize text and inspect semantic profiles ✓ How to compare sentiment diffusion with recommendation exposure ✓ How to inspect multiplex interaction layers and combined polarization ✓ How to analyze topic spread, adoption, and peak engagement ✓ How to review moderation, forum session, and summary diagnostics
Next Steps
Visualization Tutorial: Create publication-ready visualizations using ySights’ viz module