{"doi":"10.1145/2463676.2463713","title":"Petabyte scale databases and storage systems at Facebook","abstract":null,"journal":"Proceedings of the 2013 ACM SIGMOD International Conference on Management of Data","year":2013,"id":601323,"datarank":0.4493598410330987,"base_score":2.995732273553991,"endowment":2.995732273553991,"self_citation_contribution":0.4493598410330987,"citation_network_contribution":0.0,"self_endowment_contribution":0.4493598410330987,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":19,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":null,"is_data_producer":false,"deposit_databanks":null,"is_oa":false,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":null,"fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":1541850,"name":"Dhruba Borthakur","orcid":null,"position":0,"is_corresponding":false}],"reference_count":0,"raw_metadata":{"has_enrichment":true,"resolved":true,"title":"Petabyte scale databases and storage systems at Facebook","abstract":"At Facebook, we use various types of databases and storage system to satisfy the needs of different applications. The solutions built around these data store systems have a common set of requirements: they have to be highly scalable, maintenance costs should be low and they have to perform efficiently. We use a sharded mySQL+memcache solution to support real-time access of tens of petabytes of data and we use TAO to provide consistency of this web-scale database across geographical distances. We use Haystack data store for storing the 3 billion new photos we host every week. We use Apache Hadoop to mine intelligence from 100 petabytes of click logs and combine it with the power of Apache HBase to store all Facebook Messages.","is_dataset_classified":null,"base_score":2.995732273553991,"endowment":2.995732273553991,"datacite_reuse_total":0,"file_count":0,"downloads":0,"views":0,"has_version_chain":false,"is_dataset":false,"is_oa":false,"pmid":"23304386","pmcid":null,"openalex_id":"https://openalex.org/W1979756627","authors":[],"funders":[],"total_grants":0,"fwci":null,"citation_percentile":null,"influential_citations":0,"citation_trend":[{"year":2013,"count":1},{"year":2014,"count":2},{"year":2015,"count":3},{"year":2017,"count":1},{"year":2018,"count":3},{"year":2020,"count":1},{"year":2021,"count":1},{"year":2022,"count":2},{"year":2023,"count":2},{"year":2024,"count":2},{"year":2025,"count":1}],"oa_status":"closed","license":"https://www.acm.org/publications/policies/copyright_policy#Background","oa_locations":[{"url":"https://dl.acm.org/doi/10.1145/2463676.2463713","host_type":"publisher"},{"url":"https://dl.acm.org/doi/pdf/10.1145/2463676.2463713","host_type":"publisher"},{"url":"https://doi.org/10.1145/2463676.2463713","host_type":""}],"fields_of_study":["Cloud Computing and Resource Management","Web Data Mining and Analysis","Data Management and Algorithms"],"mesh_terms":[],"keywords":["Petabyte","Scalability","Computer science","Database","Haystack","NoSQL","Consistency (knowledge bases)","Scale (ratio)","World Wide Web","Big data","Operating system"],"sdg_mappings":[],"linked_datasets":[],"clinical_trials":[],"software_tools":[],"database_accessions":[],"source":"live","citation_network_status":"fetched"},"created_at":"2026-07-29T15:52:55.760359Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}