{"data":{"slug":"apache-spark","name":"Apache Spark","tagline":"","homepage":"https://spark.apache.org","category":{"slug":"data-engineering","name":"Data Engineering Tools"},"vendor":null,"score":{"composite":51.9,"confidence":0.35,"confidenceBand":"low","breakdown":{"capabilities":{"value":45.8,"weight":0.078,"present":true,"contribution":3.6},"repo_stars":{"value":87.6,"weight":0.026,"present":true,"contribution":2.3},"integrations":{"value":22.4,"weight":0.07,"present":true,"contribution":1.6},"dependent_projects":{"value":65.7,"weight":0.063,"present":true,"contribution":4.1},"dev_activity":{"value":63.1,"weight":0.094,"present":true,"contribution":5.9},"release_cadence":{"value":0,"weight":0.052,"present":false,"contribution":0},"security_posture":{"value":5,"weight":0.073,"present":false,"contribution":0},"package_downloads":{"value":0,"weight":0.136,"present":false,"contribution":0},"security_score":{"value":56,"weight":0.042,"present":true,"contribution":2.4},"community_qa_activity":{"value":0,"weight":0.063,"present":false,"contribution":0}},"computedAt":"2026-09-10T17:41:40.184Z","stale":false},"uri":"https://www.vioscale.ai/software/apache-spark","aliases":["apache-spark"],"status":"published","crawlStatus":"ok","indexStatus":"no_facts","nextCrawlAt":null,"categories":[{"slug":"data-engineering","name":"Data Engineering Tools","isPrimary":true}],"facts":{"activity":[{"attribute":"activity.commits_last_30d","value":100,"provenance":{"source":"https://github.com/apache/spark/pulse","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.65}}],"adoption":[{"attribute":"adoption.github_stars","value":43974,"provenance":{"source":"https://github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.9}},{"attribute":"adoption.dependent_repos","value":8772,"provenance":{"source":"https://packages.ecosyste.ms/api/v1/packages/lookup?repository_url=https%3A%2F%2Fgithub.com%2Fapache%2Fspark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.697Z","confidence":0.85}}],"deployment":[{"attribute":"deployment.options","value":{"on_prem":true,"self_hosted":true},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}}],"language":[{"attribute":"language.primary","value":"Scala","provenance":{"source":"https://github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.9}}],"license":[{"attribute":"license.spdx","value":"Apache-2.0","provenance":{"source":"https://github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.95}}],"platform":[{"attribute":"platform.support","value":{"cli":true},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}}],"pricing":[{"attribute":"pricing","value":{"type":"open_source","freeTier":true,"sourceUrl":"https://spark.apache.org","retrievedAt":"2026-08-14T13:48:09.654Z"},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing.free_tier","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing.model","value":"commercial","provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.688Z","confidence":0.4}},{"attribute":"pricing.price_level","value":"free","provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing.transparent","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6}}],"features":[{"attribute":"features.capabilities","value":{"role":"engine","paradigm":"both","self_hosted_oss":true,"distributed_executor":true},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}}],"description":[{"attribute":"description.long","value":"Apache Spark is an open-source, multi-language distributed computing engine that unifies data engineering, data science, and machine learning workloads. It processes data at scale using batch or streaming paradigms, provides SQL query capabilities for analytics, and includes built-in libraries for machine learning and graph processing.","provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}}],"integrations":[{"attribute":"integrations.count","value":5,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"integrations.list","value":[{"name":"Hadoop"},{"name":"HDFS"},{"name":"YARN"},{"name":"Kubernetes"},{"name":"Docker"}],"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}}],"market":[{"attribute":"market.availability","value":{"primaryMarkets":[],"availabilityScope":"global","availableCountries":[],"notAvailableCountries":[]},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.75}}],"security":[{"attribute":"security.disclosure_policy","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"security.scorecard","value":5.6,"provenance":{"source":"https://api.securityscorecards.dev/projects/github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.703Z","confidence":0.9}},{"attribute":"security.vulnerabilities","value":{"count":9,"source":"https://advisories.ecosyste.ms/api/v1/advisories?ecosystem=maven&package_name=org.apache.spark%3Aspark-core_2.11&per_page=100","last_12m":1,"max_severity":"CRITICAL"},"provenance":{"source":"https://advisories.ecosyste.ms/api/v1/advisories?ecosystem=maven&package_name=org.apache.spark%3Aspark-core_2.11&per_page=100","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.720Z","confidence":0.9}}],"content":[{"attribute":"content.faq","value":[{"answer":"Apache Spark is an open-source, multi-language distributed computing engine that unifies data engineering, data science, and machine learning workloads. It processes data at scale using batch or streaming paradigms, provides SQL query capabilities for analytics, and includes built-in libraries for machine learning and graph processing. It is indexed under Data Engineering Tools.","source":"https://spark.apache.org","question":"What is Apache Spark?","confidence":0.6},{"answer":"Apache Spark is open source, so it can be self-hosted and used at no licence cost. It is released under the Apache-2.0 licence. Pricing changes often, so verify at source before relying on it.","source":"https://spark.apache.org","question":"Is Apache Spark free to use?","confidence":0.6},{"answer":"Apache Spark supports a command-line interface. Platforms we have not confirmed are simply not listed here rather than ruled out.","source":"https://spark.apache.org","question":"What platforms does Apache Spark support?","confidence":0.6},{"answer":"Yes. Apache Spark can be deployed on-premise and self-hosted, so it does not have to run on the vendor's infrastructure.","source":"https://spark.apache.org","question":"Can Apache Spark be self-hosted?","confidence":0.6},{"answer":"We have confirmed 5 integrations for Apache Spark, including Hadoop, HDFS, YARN, Kubernetes and Docker. This is what we could verify from public sources, so the vendor may support others we have not indexed.","source":"https://spark.apache.org","question":"What does Apache Spark integrate with?","confidence":0.6},{"answer":"Yes. Apache Spark is published under the Apache-2.0 licence, a permissive licence that generally allows commercial use and modification. Licence terms can change between releases, so verify against the repository for the version you intend to use.","source":"https://github.com/apache/spark","question":"Is Apache Spark open source?","confidence":0.95}],"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T10:20:52.151Z","confidence":0.65833336}}]},"factList":[{"attribute":"activity.commits_last_30d","value":100,"provenance":{"source":"https://github.com/apache/spark/pulse","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.65}},{"attribute":"adoption.github_stars","value":43974,"provenance":{"source":"https://github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.9}},{"attribute":"deployment.options","value":{"on_prem":true,"self_hosted":true},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"language.primary","value":"Scala","provenance":{"source":"https://github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.9}},{"attribute":"license.spdx","value":"Apache-2.0","provenance":{"source":"https://github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.674Z","confidence":0.95}},{"attribute":"platform.support","value":{"cli":true},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing","value":{"type":"open_source","freeTier":true,"sourceUrl":"https://spark.apache.org","retrievedAt":"2026-08-14T13:48:09.654Z"},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing.free_tier","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing.model","value":"commercial","provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.688Z","confidence":0.4}},{"attribute":"pricing.price_level","value":"free","provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"pricing.transparent","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6}},{"attribute":"features.capabilities","value":{"role":"engine","paradigm":"both","self_hosted_oss":true,"distributed_executor":true},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"description.long","value":"Apache Spark is an open-source, multi-language distributed computing engine that unifies data engineering, data science, and machine learning workloads. It processes data at scale using batch or streaming paradigms, provides SQL query capabilities for analytics, and includes built-in libraries for machine learning and graph processing.","provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"integrations.count","value":5,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"integrations.list","value":[{"name":"Hadoop"},{"name":"HDFS"},{"name":"YARN"},{"name":"Kubernetes"},{"name":"Docker"}],"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"market.availability","value":{"primaryMarkets":[],"availabilityScope":"global","availableCountries":[],"notAvailableCountries":[]},"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.75}},{"attribute":"security.disclosure_policy","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-08-14T13:48:09.654Z","confidence":0.6}},{"attribute":"adoption.dependent_repos","value":8772,"provenance":{"source":"https://packages.ecosyste.ms/api/v1/packages/lookup?repository_url=https%3A%2F%2Fgithub.com%2Fapache%2Fspark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.697Z","confidence":0.85}},{"attribute":"security.scorecard","value":5.6,"provenance":{"source":"https://api.securityscorecards.dev/projects/github.com/apache/spark","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.703Z","confidence":0.9}},{"attribute":"security.vulnerabilities","value":{"count":9,"source":"https://advisories.ecosyste.ms/api/v1/advisories?ecosystem=maven&package_name=org.apache.spark%3Aspark-core_2.11&per_page=100","last_12m":1,"max_severity":"CRITICAL"},"provenance":{"source":"https://advisories.ecosyste.ms/api/v1/advisories?ecosystem=maven&package_name=org.apache.spark%3Aspark-core_2.11&per_page=100","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T17:38:27.720Z","confidence":0.9}},{"attribute":"content.faq","value":[{"answer":"Apache Spark is an open-source, multi-language distributed computing engine that unifies data engineering, data science, and machine learning workloads. It processes data at scale using batch or streaming paradigms, provides SQL query capabilities for analytics, and includes built-in libraries for machine learning and graph processing. It is indexed under Data Engineering Tools.","source":"https://spark.apache.org","question":"What is Apache Spark?","confidence":0.6},{"answer":"Apache Spark is open source, so it can be self-hosted and used at no licence cost. It is released under the Apache-2.0 licence. Pricing changes often, so verify at source before relying on it.","source":"https://spark.apache.org","question":"Is Apache Spark free to use?","confidence":0.6},{"answer":"Apache Spark supports a command-line interface. Platforms we have not confirmed are simply not listed here rather than ruled out.","source":"https://spark.apache.org","question":"What platforms does Apache Spark support?","confidence":0.6},{"answer":"Yes. Apache Spark can be deployed on-premise and self-hosted, so it does not have to run on the vendor's infrastructure.","source":"https://spark.apache.org","question":"Can Apache Spark be self-hosted?","confidence":0.6},{"answer":"We have confirmed 5 integrations for Apache Spark, including Hadoop, HDFS, YARN, Kubernetes and Docker. This is what we could verify from public sources, so the vendor may support others we have not indexed.","source":"https://spark.apache.org","question":"What does Apache Spark integrate with?","confidence":0.6},{"answer":"Yes. Apache Spark is published under the Apache-2.0 licence, a permissive licence that generally allows commercial use and modification. Licence terms can change between releases, so verify against the repository for the version you intend to use.","source":"https://github.com/apache/spark","question":"Is Apache Spark open source?","confidence":0.95}],"provenance":{"source":"https://spark.apache.org","sourceType":"vioscale-crawler","retrievedAt":"2026-09-10T10:20:52.151Z","confidence":0.65833336}}],"faq":[{"answer":"Apache Spark is an open-source, multi-language distributed computing engine that unifies data engineering, data science, and machine learning workloads. It processes data at scale using batch or streaming paradigms, provides SQL query capabilities for analytics, and includes built-in libraries for machine learning and graph processing. It is indexed under Data Engineering Tools.","source":"https://spark.apache.org","question":"What is Apache Spark?","confidence":0.6},{"answer":"Apache Spark is open source, so it can be self-hosted and used at no licence cost. It is released under the Apache-2.0 licence. Pricing changes often, so verify at source before relying on it.","source":"https://spark.apache.org","question":"Is Apache Spark free to use?","confidence":0.6},{"answer":"Apache Spark supports a command-line interface. Platforms we have not confirmed are simply not listed here rather than ruled out.","source":"https://spark.apache.org","question":"What platforms does Apache Spark support?","confidence":0.6},{"answer":"Yes. Apache Spark can be deployed on-premise and self-hosted, so it does not have to run on the vendor's infrastructure.","source":"https://spark.apache.org","question":"Can Apache Spark be self-hosted?","confidence":0.6},{"answer":"We have confirmed 5 integrations for Apache Spark, including Hadoop, HDFS, YARN, Kubernetes and Docker. This is what we could verify from public sources, so the vendor may support others we have not indexed.","source":"https://spark.apache.org","question":"What does Apache Spark integrate with?","confidence":0.6},{"answer":"Yes. Apache Spark is published under the Apache-2.0 licence, a permissive licence that generally allows commercial use and modification. Licence terms can change between releases, so verify against the repository for the version you intend to use.","source":"https://github.com/apache/spark","question":"Is Apache Spark open source?","confidence":0.95}],"faqSource":"generated","market":{"availabilityScope":"global","availableCountries":[],"notAvailableCountries":[],"primaryMarkets":[]},"integrations":[{"name":"Hadoop"},{"name":"HDFS"},{"name":"YARN"},{"name":"Kubernetes"},{"name":"Docker"}],"vulnerabilities":{"count":9,"last12m":1,"maxSeverity":"CRITICAL","source":"https://advisories.ecosyste.ms/api/v1/advisories?ecosystem=maven&package_name=org.apache.spark%3Aspark-core_2.11&per_page=100"},"claimed":false,"updatedAt":"2026-09-10T18:00:00.399Z","formats":{"json":"https://www.vioscale.ai/api/v1/software/apache-spark","markdown":"https://www.vioscale.ai/software/apache-spark.md","html":"https://www.vioscale.ai/software/apache-spark"}},"meta":{"source":"https://www.vioscale.ai","license":"CC-BY-4.0","generatedAt":"2026-09-21T01:30:17.817Z","disclaimer":"Independent, evidence-based. Every fact carries provenance."}}