Algorithms in ySights

This tutorial covers the analytical algorithms available in ySights for understanding simulation dynamics.

What You’ll Learn

  • Profile similarity analysis

  • Semantic text enrichment and similarity checks

  • Sentiment diffusion and recommendation exposure

  • Topic lifecycle analysis

  • Multiplex interaction diagnostics

  • Moderation and forum session summaries


[ ]:
from pathlib import Path

from ysights import YDataHandler
from ysights.algorithms import (
    profile_topics_similarity,
    visibility_paradox,
    user_visibility_vs_neighbors,
    visibility_paradox_population_size_null,
    engagement_momentum,
    personalization_balance_score,
    sentiment_diffusion_metrics,
)
from ysights.algorithms.topics import topic_spread, adoption_rate, peak_engagement_time
import matplotlib.pyplot as plt
import numpy as np

plt.style.use('seaborn-v0_8-darkgrid')
%matplotlib inline

[ ]:
# Initialize data handler and get network
from pathlib import Path


def resolve_example_db():
    candidates = [
        Path("ysocial_db.db"),
        Path("../notebooks/ysocial_db.db"),
        Path("../../notebooks/ysocial_db.db"),
        Path("docs/notebooks/ysocial_db.db"),
    ]
    for candidate in candidates:
        if candidate.exists():
            return str(candidate.resolve())
    return "ysocial_db.db"

db_path = resolve_example_db()
ydh = YDataHandler(db_path)
network = ydh.social_network()

1. Profile Similarity Analysis

Analyzes how similar users’ interest profiles are across the network.

[ ]:
similarity_scores = profile_topics_similarity(ydh, network)
similarity_values = list(similarity_scores.values())

print(f"Computed {len(similarity_scores)} similarity scores")
print(f"\nSimilarity Statistics:")
print(f"  Mean: {np.mean(similarity_values):.4f}")
print(f"  Median: {np.median(similarity_values):.4f}")
print(f"  Std Dev: {np.std(similarity_values):.4f}")
print(f"  Min: {min(similarity_values):.4f}")
print(f"  Max: {max(similarity_values):.4f}")
[ ]:
plt.figure(figsize=(12, 5))

plt.subplot(1, 2, 1)
plt.hist(similarity_values, bins=50, edgecolor='black', alpha=0.7, color='coral')
plt.xlabel('Similarity Score', fontsize=11)
plt.ylabel('Frequency', fontsize=11)
plt.title('Profile Similarity Distribution', fontsize=13, fontweight='bold')
plt.axvline(np.mean(similarity_values), color='red', linestyle='--', label=f'Mean: {np.mean(similarity_values):.3f}')
plt.legend()
plt.grid(True, alpha=0.3)

plt.subplot(1, 2, 2)
plt.boxplot(similarity_values, vert=True)
plt.ylabel('Similarity Score', fontsize=11)
plt.title('Profile Similarity Box Plot', fontsize=13, fontweight='bold')
plt.grid(True, alpha=0.3, axis='y')

plt.tight_layout()
plt.show()

2. Recommendation System Metrics

Engagement Momentum

Measures how consistently users engage with recommended content over time.

[ ]:
momentum = engagement_momentum(ydh, time_window_rounds=24)

print("Engagement Momentum Analysis:")
print(f"  Users analyzed: {len(momentum)}")
print(f"  Average momentum: {np.mean(list(momentum.values())):.4f}")
print(f"  Median momentum: {np.median(list(momentum.values())):.4f}")

top_momentum = sorted(momentum.items(), key=lambda x: x[1], reverse=True)[:5]
print("\nTop 5 Users by Engagement Momentum:")
for i, (user, score) in enumerate(top_momentum, 1):
    print(f"  {i}. User {user}: {score:.4f}")

Personalization Balance Score

Measures how well the recommendation system balances exploration vs. exploitation.

[ ]:
balance_scores = personalization_balance_score(ydh)

print("Personalization Balance Analysis:")
print(f"  Users analyzed: {len(balance_scores)}")
print(f"  Average balance: {np.mean(list(balance_scores.values())):.4f}")
print(f"  Median balance: {np.median(list(balance_scores.values())):.4f}")
print("\nInterpretation:")
print("  Score → 0: Heavy exploitation (narrow recommendations)")
print("  Score → 1: Heavy exploration (diverse recommendations)")
[ ]:
plt.figure(figsize=(10, 6))
plt.hist(list(balance_scores.values()), bins=30, edgecolor='black', alpha=0.7, color='seagreen')
plt.xlabel('Personalization Balance Score', fontsize=12)
plt.ylabel('Number of Users', fontsize=12)
plt.title('Distribution of Personalization Balance', fontsize=14, fontweight='bold')
plt.axvline(0.5, color='red', linestyle='--', linewidth=2, label='Perfect Balance')
plt.legend()
plt.grid(True, alpha=0.3)
plt.tight_layout()
plt.show()

3. Topic Lifecycle Analysis

These helpers are thin wrappers around the implemented topic lifecycle analysis in YDataHandler.

[ ]:
lifecycles = topic_spread(ydh)
adoption_rates = adoption_rate(ydh)
peak_periods = peak_engagement_time(ydh)

topic_ids = list(lifecycles.keys())
print(f"Topics analyzed: {len(topic_ids)}")

for topic_id in topic_ids[:5]:
    lifecycle = lifecycles[topic_id]
    print(
        f"  Topic {topic_id}: posts={lifecycle['post_count']}, "
        f"authors={lifecycle['author_count']}, "
        f"peak={lifecycle['peak_period']}, "
        f"adoption={adoption_rates[topic_id]:.3f}"
    )

if topic_ids:
    topic_id = topic_ids[0]
    print(f"\nTimeline preview for topic {topic_id}:")
    print(lifecycles[topic_id]["timeline"].head().to_string(index=False))

4. Semantic and Content Diagnostics

These helpers expose lightweight semantic features, text normalization, and similarity checks that work across content sources.

Text normalization and semantic profiles

Normalize a post, inspect its semantic profile, and compare it with a short sample string.

[ ]:
agent_id = next(iter(ydh.agent_mapping()))
sample_posts = ydh.posts_by_agent(agent_id).get_posts()

if sample_posts:
    sample_post = sample_posts[0]
    post_profile = ydh.post_semantic_profile(sample_post.id)

    print(f"Semantic profile for post {sample_post.id}:")
    for key in [
        "token_count",
        "word_count",
        "lexical_diversity",
        "readability_proxy",
        "punctuation_intensity",
    ]:
        print(f"  {key}: {post_profile[key]}")

    similarity = ydh.semantic_similarity(
        sample_post.text,
        "Short reply about a similar topic",
    )
    print("\nSemantic similarity against a short sample:")
    print(f"  mode: {similarity['mode']}")
    print(f"  score: {similarity['score']:.4f}")
    print(f"  token_jaccard: {similarity['token_jaccard']:.4f}")
else:
    print("No sample posts were found in this dataset.")

Exercise

Compare two short content snippets and inspect how lexical similarity changes when the wording changes only slightly.

[ ]:
exercise_similarity = ydh.semantic_similarity(
    "Forum users share updates",
    "Users share updates in the forum",
)
print("Exercise result:")
print(f"  mode: {exercise_similarity['mode']}")
print(f"  score: {exercise_similarity['score']:.4f}")
print(f"  token_jaccard: {exercise_similarity['token_jaccard']:.4f}")

Sentiment diffusion and recommendation exposure

Summarize how sentiment propagates through recommendations and how often the same exposure pairs recur.

[ ]:
sentiment = sentiment_diffusion_metrics(ydh)
exposure = ydh.recommendation_exposure_summary()

print("Sentiment diffusion:")
for key in ["available", "post_count", "labeled_post_count", "sentiment_coverage", "recommendation_count"]:
    if key in sentiment:
        print(f"  {key}: {sentiment[key]}")
print("  sentiment_label_counts:", sentiment["sentiment_label_counts"])
print("  average_scores:", sentiment["average_scores"])
print("\nSentiment diffusion timeline preview:")
print(sentiment["timeline"].head().to_string(index=False))

print("\nRecommendation exposure summary:")
for key in ["exposure_count", "unique_recipients", "unique_posts", "unique_authors"]:
    print(f"  {key}: {exposure[key]}")
print("  conversion_rates:", exposure["conversion_rates"])
print("  feedback_loop:", exposure["feedback_loop"])
print("\nExposure timeline preview:")
print(exposure["timeline"].head().to_string(index=False))

Multiplex interaction overview

Inspect the available interaction layers and the dominant nodes in each layer.

[ ]:
multiplex = ydh.multiplex_metrics()

print(f"Available layers: {multiplex['available_layers']}")
for layer_name, layer_metrics in multiplex["layer_metrics"].items():
    print(f"  {layer_name}: {layer_metrics['edge_count']} edges, weight_sum={layer_metrics['weight_sum']}")
    centrality = layer_metrics["centrality"]
    print(f"    top in-degree node: {centrality['top_in_degree']['node']}")
    print(f"    strong tie share: {layer_metrics['tie_strength']['strong_tie_share']:.4f}")

print("\nPairwise overlap preview:")
for pair_name, overlap in list(multiplex["pairwise_overlap"].items())[:3]:
    print(
        f"  {pair_name}: edge_jaccard={overlap['edge_jaccard']:.4f}, "
        f"node_jaccard={overlap['node_jaccard']:.4f}"
    )

print("\nCombined polarization:")
print(multiplex["combined_polarization"])

5. Operational and Moderation Summaries

These helpers are useful when you want a compact dataset overview before running a deeper analysis.

[ ]:
summary_report = ydh.summary_report()
summary_frame = ydh.summary_frame()
moderation_summary = ydh.moderation_summary()
forum_sessions = ydh.forum_session_summaries()
interaction_layers = ydh.interaction_layers()
multiplex = ydh.multiplex_metrics()
cache_info = ydh.analysis_cache_info()
recommended_indexes = ydh.recommended_indexes()
benchmark = ydh.benchmark_analytics(iterations=1)

print("Summary Report (selected keys):")
for key in ["post_count", "thread_count", "report_count", "forum_session_count", "moderated_posts"]:
    if key in summary_report:
        print(f"  {key}: {summary_report[key]}")

print("\nInteraction Layer Summary:")
print(f"  layers: {list(interaction_layers.keys())}")
print(f"  combined edges: {multiplex['combined']['edge_count']}")
print(f"  reciprocity: {multiplex['combined_polarization']['reciprocity']}")

print("\nSummary Frame Preview:")
print(summary_frame.head().to_string(index=False))

print("\nModeration Summary:")
print(moderation_summary)

print("\nForum Session Summaries:")
print(forum_sessions)

print("\nCache Diagnostics:")
print(cache_info)

print("\nRecommended Indexes:")
print(recommended_indexes)

print("\nBenchmark Targets:")
print(list(benchmark["metrics"].keys()))

Summary

In this tutorial, you learned:

✓ How to measure profile similarity across the network ✓ How to normalize text and inspect semantic profiles ✓ How to compare sentiment diffusion with recommendation exposure ✓ How to inspect multiplex interaction layers and combined polarization ✓ How to analyze topic spread, adoption, and peak engagement ✓ How to review moderation, forum session, and summary diagnostics

Next Steps

  • Visualization Tutorial: Create publication-ready visualizations using ySights’ viz module