{"data":{"slug":"apache-spark","name":"Apache Spark","tagline":"","homepage":"https://spark.apache.org","category":{"slug":"data-engineering","name":"Data Engineering Tools"},"vendor":null,"score":{"composite":73.5,"confidence":0.16,"confidenceBand":"low","breakdown":{"price_level":{"value":100,"weight":0.058,"present":true,"contribution":5.8},"reliability":{"value":0,"weight":0.082,"present":false,"contribution":0},"capabilities":{"value":57.8,"weight":0.094,"present":true,"contribution":5.4},"github_stars":{"value":87.6,"weight":0.029,"present":true,"contribution":2.5},"integrations":{"value":0,"weight":0.175,"present":false,"contribution":0},"github_activity":{"value":63.1,"weight":0.105,"present":true,"contribution":6.6},"release_cadence":{"value":0,"weight":0.058,"present":false,"contribution":0},"security_posture":{"value":0,"weight":0.082,"present":false,"contribution":0},"package_downloads":{"value":0,"weight":0.152,"present":false,"contribution":0},"pricing_transparency":{"value":80,"weight":0.094,"present":true,"contribution":7.5},"stackoverflow_activity":{"value":0,"weight":0.07,"present":false,"contribution":0}},"computedAt":"2026-08-05T14:30:23.009Z"},"uri":"https://www.vioscale.ai/software/apache-spark","aliases":["apache-spark"],"status":"active","facts":{"activity":[{"attribute":"activity.commits_last_30d","value":100,"provenance":{"source":"https://github.com/apache/spark/pulse","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.65,"snapshotRef":null}}],"adoption":[{"attribute":"adoption.github_stars","value":43759,"provenance":{"source":"https://github.com/apache/spark","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.9,"snapshotRef":null}}],"deployment":[{"attribute":"deployment.options","value":{"cloud":true,"on_prem":true,"self_hosted":true},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}}],"language":[{"attribute":"language.primary","value":"Scala","provenance":{"source":"https://github.com/apache/spark","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.9,"snapshotRef":null}}],"license":[{"attribute":"license.spdx","value":"Apache-2.0","provenance":{"source":"https://github.com/apache/spark","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.95,"snapshotRef":null}}],"platform":[{"attribute":"platform.support","value":{"cli":true,"linux":true},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}}],"pricing":[{"attribute":"pricing","value":{"type":"open_source","freeTier":true,"sourceUrl":"https://spark.apache.org","retrievedAt":"2026-08-05T14:25:05.367Z"},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.free_tier","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.model","value":"commercial","provenance":{"source":"https://spark.apache.org","sourceType":"crawler:pricing","retrievedAt":"2026-08-05T14:22:01.677Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.price_level","value":"free","provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.transparent","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}}],"features":[{"attribute":"features.capabilities","value":{"role":"orchestrator","paradigm":"both","python_first":true,"managed_cloud":false,"self_hosted_oss":true,"distributed_executor":true},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}}],"description":[{"attribute":"description.long","value":"Apache Spark is a multi-language engine for executing data engineering, data science, and machine learning on single-node machines or clusters. It unifies batch and real-time streaming processing with SQL analytics and machine learning capabilities.","provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-01T14:18:22.165Z","confidence":0.6,"snapshotRef":null}}],"market":[{"attribute":"market.availability","value":{"hqCountry":"US","primaryMarkets":[],"availabilityScope":"global","availableCountries":[],"notAvailableCountries":[]},"provenance":{"source":"https://spark.apache.org","sourceType":"crawler:website","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.75,"snapshotRef":null}}]},"factList":[{"attribute":"activity.commits_last_30d","value":100,"provenance":{"source":"https://github.com/apache/spark/pulse","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.65,"snapshotRef":null}},{"attribute":"adoption.github_stars","value":43759,"provenance":{"source":"https://github.com/apache/spark","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.9,"snapshotRef":null}},{"attribute":"deployment.options","value":{"cloud":true,"on_prem":true,"self_hosted":true},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"language.primary","value":"Scala","provenance":{"source":"https://github.com/apache/spark","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.9,"snapshotRef":null}},{"attribute":"license.spdx","value":"Apache-2.0","provenance":{"source":"https://github.com/apache/spark","sourceType":"crawler:github","retrievedAt":"2026-08-01T14:16:54.631Z","confidence":0.95,"snapshotRef":null}},{"attribute":"platform.support","value":{"cli":true,"linux":true},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing","value":{"type":"open_source","freeTier":true,"sourceUrl":"https://spark.apache.org","retrievedAt":"2026-08-05T14:25:05.367Z"},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.free_tier","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.model","value":"commercial","provenance":{"source":"https://spark.apache.org","sourceType":"crawler:pricing","retrievedAt":"2026-08-05T14:22:01.677Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.price_level","value":"free","provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"pricing.transparent","value":true,"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"features.capabilities","value":{"role":"orchestrator","paradigm":"both","python_first":true,"managed_cloud":false,"self_hosted_oss":true,"distributed_executor":true},"provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.6,"snapshotRef":null}},{"attribute":"description.long","value":"Apache Spark is a multi-language engine for executing data engineering, data science, and machine learning on single-node machines or clusters. It unifies batch and real-time streaming processing with SQL analytics and machine learning capabilities.","provenance":{"source":"https://spark.apache.org","sourceType":"assisted:claude-haiku-4-5","retrievedAt":"2026-08-01T14:18:22.165Z","confidence":0.6,"snapshotRef":null}},{"attribute":"market.availability","value":{"hqCountry":"US","primaryMarkets":[],"availabilityScope":"global","availableCountries":[],"notAvailableCountries":[]},"provenance":{"source":"https://spark.apache.org","sourceType":"crawler:website","retrievedAt":"2026-08-05T14:25:05.367Z","confidence":0.75,"snapshotRef":null}}],"market":{"availabilityScope":"global","availableCountries":[],"notAvailableCountries":[],"primaryMarkets":[],"hqCountry":"US"},"claimed":false,"updatedAt":"2026-08-05T14:30:23.186Z","formats":{"json":"https://www.vioscale.ai/api/v1/software/apache-spark","markdown":"https://www.vioscale.ai/software/apache-spark.md","html":"https://www.vioscale.ai/software/apache-spark"}},"meta":{"source":"https://www.vioscale.ai","license":"CC-BY-4.0","generatedAt":"2026-08-06T23:52:58.922Z","disclaimer":"Independent, evidence-based. Every fact carries provenance."}}