{
  "@context": "https://schema.org",
  "@type": "Article",
  "@id": "https://www.vidyasource.com/case-studies/machine-learning-threat-detection-trss/",
  "url": "https://www.vidyasource.com/case-studies/machine-learning-threat-detection-trss/",
  "mainEntityOfPage": "https://www.vidyasource.com/case-studies/machine-learning-threat-detection-trss/",
  "headline": "TRSS Turns Threat Detection Into a Named Product Line",
  "description": "Thomson Reuters Special Services grew threat detection into a product line. A machine learning threat detection case study on Kafka, Spark, and Scala.",
  "about": {
    "@type": "Organization",
    "name": "Thomson Reuters Special Services (TRSS)"
  },
  "author": {
    "@id": "https://www.vidyasource.com/#organization"
  },
  "publisher": {
    "@id": "https://www.vidyasource.com/#organization"
  },
  "image": "https://www.vidyasource.com/img/blog/big-data.jpg",
  "keywords": [
    "Machine Learning",
    "Data",
    "Architecture",
    "AI",
    "Scala",
    "Python",
    "Apache Kafka",
    "Apache Spark",
    "Spark MLlib",
    "MongoDB",
    "Play Framework",
    "Spring Integration",
    "Akka",
    "Angular",
    "Microsoft Azure"
  ],
  "mainEntity": [
    {
      "@type": "Question",
      "name": "What did Vidya build for Thomson Reuters Special Services?",
      "acceptedAnswer": {
        "@type": "Answer",
        "text": "Vidya joined a team of senior engineers on the TRSS threat-detection application and re-architected it into a streaming machine learning pipeline. Spring Integration normalized data from multiple social media and online APIs and published it to Apache Kafka. A Scala model on Apache Spark and Spark MLlib applied natural language processing to that stream and wrote threat scores into MongoDB for the Play Framework interface to serve."
      }
    },
    {
      "@type": "Question",
      "name": "How does a real-time entity resolution and fraud detection pipeline stay fast under load?",
      "acceptedAnswer": {
        "@type": "Answer",
        "text": "The pipeline moves the analytical work off the request path. Vidya normalized every incoming feed into one message format, published it to a Kafka topic, and let the model score continuously against that stream. The user interface then reads a materialized view, which means response time depends on a lookup rather than on how much data arrived that morning."
      }
    },
    {
      "@type": "Question",
      "name": "Why Kafka and Spark instead of the Lambda Architecture with Hadoop?",
      "acceptedAnswer": {
        "@type": "Answer",
        "text": "Vidya published an early plan to evolve the platform toward Nathan Marz's Lambda Architecture with Hadoop and Elasticsearch, then changed course. Lambda asks a team to maintain a batch layer and a speed layer that each implement the same scoring rules, and TRSS engineers would have carried that duplication for the life of the product. Kafka and Spark gave them one code path, at the cost of replaying the log to recompute history."
      }
    },
    {
      "@type": "Question",
      "name": "Does Vidya build big data risk scoring platforms for federal customers today?",
      "acceptedAnswer": {
        "@type": "Answer",
        "text": "Yes. The TRSS engagement is Vidya's machine learning threat-detection reference, and Vidya has carried the same event-driven data architecture into federal work, including the predictive analytics platform on Azure Databricks that Vidya architected under Consular Systems Modernization at the Department of State. Vidya holds GSA MAS contract 47QTCA24D0069 for that work."
      }
    }
  ]
}