diff --git a/chunker_test_results.txt b/chunker_test_results.txt new file mode 100644 index 00000000..a4e42b57 --- /dev/null +++ b/chunker_test_results.txt @@ -0,0 +1,55 @@ +============================= test session starts ============================== +platform darwin -- Python 3.10.12, pytest-8.4.2, pluggy-1.6.0 -- /Users/diyakothari/Desktop/Wattbot/venv/bin/python +cachedir: .pytest_cache +rootdir: /Users/diyakothari/Desktop/Wattbot +plugins: cov-7.0.0 +collecting ... collected 24 items + +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_text_basic PASSED [ 4%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_text_overlap PASSED [ 8%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_text_empty PASSED [ 12%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_text_shorter_than_chunk_size PASSED [ 16%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_text_exact_chunk_size PASSED [ 20%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_text_whitespace_handling PASSED [ 25%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_custom_chunk_size PASSED [ 29%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_custom_overlap PASSED [ 33%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_zero_overlap PASSED [ 37%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_large_overlap PASSED [ 41%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_multiline_text PASSED [ 45%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_extract_text_from_pdf_real_file PASSED [ 50%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_extract_text_from_pdf_returns_string PASSED [ 54%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_extract_text_from_pdf_mock PASSED [ 58%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_extract_tables_from_pdf_success PASSED [ 62%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_extract_tables_from_pdf_failure PASSED [ 66%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_extract_tables_both_flavors PASSED [ 70%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_pdf_real_file PASSED [ 75%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_pdf_creates_text_chunks PASSED [ 79%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_pdf_creates_table_chunks PASSED [ 83%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_save_chunks_to_json PASSED [ 87%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_save_chunks_creates_directory PASSED [ 91%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_init_default_parameters PASSED [ 95%] +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_init_custom_parameters PASSED [100%] + +=============================== warnings summary =============================== +venv/lib/python3.10/site-packages/PyPDF2/__init__.py:21 + /Users/diyakothari/Desktop/Wattbot/venv/lib/python3.10/site-packages/PyPDF2/__init__.py:21: DeprecationWarning: PyPDF2 is deprecated. Please move to the pypdf library instead. + warnings.warn( + +venv/lib/python3.10/site-packages/pypdf/_crypt_providers/_cryptography.py:32 + /Users/diyakothari/Desktop/Wattbot/venv/lib/python3.10/site-packages/pypdf/_crypt_providers/_cryptography.py:32: CryptographyDeprecationWarning: ARC4 has been moved to cryptography.hazmat.decrepit.ciphers.algorithms.ARC4 and will be removed from cryptography.hazmat.primitives.ciphers.algorithms in 48.0.0. + from cryptography.hazmat.primitives.ciphers.algorithms import AES, ARC4 + +wattbot2025/tests/test_chunker.py::TestDocumentChunker::test_chunk_pdf_real_file + /Users/diyakothari/Desktop/Wattbot/venv/lib/python3.10/site-packages/camelot/parsers/base.py:238: UserWarning: No tables found in table area (97.691, 188.51954679999997, 515.7449530889999, 485.81292525945946) + cols, rows, v_s, h_s = self._generate_columns_and_rows(bbox, user_cols) + +-- Docs: https://docs.pytest.org/en/stable/how-to/capture-warnings.html +================================ tests coverage ================================ +______________ coverage: platform darwin, python 3.10.12-final-0 _______________ + +Name Stmts Miss Cover Missing +--------------------------------------------------------------- +wattbot2025/src/data/chunker.py 60 7 88% 82-90 +--------------------------------------------------------------- +TOTAL 60 7 88% +======================== 24 passed, 3 warnings in 9.82s ======================== diff --git a/wattbot2025/data/ranked/amazon2023_chunks_ranked.json b/wattbot2025/data/ranked/amazon2023_chunks_ranked.json new file mode 100644 index 00000000..64cc10d9 --- /dev/null +++ b/wattbot2025/data/ranked/amazon2023_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 9.132716887768051, + "content": "ctively contribute more than 50% of emissions globally \nto Amazon’s Scope 3 footprint, to provide a plan for how \nthey will decarbonize their operations and demonstrate real \nprogress over time. We will prioritize our business toward \nthose who provide their plans and results on their path to \nnet-zero carbon emissions. In addition, we also launched our \n“Amazon Sustainability Exchange”—a free, pu", + "type": "text" + }, + { + "rank": 2, + "score": 9.096112610125404, + "content": "lity Report Appendix People\nOverview Environment Value Chain\nWater Waste and Circularity Packaging Carbon-Free Energy Carbon\n\nOur Progress\nAmazon’s Carbon Footprint\nIn 2023, our absolute carbon emission s decreased by 3%.2 \nThis overall decrease was driven by an 11% reduction in \nemissions from electricity (Scope 2) and a 5% decrease in \nindirect and supply chain emissions (Scope 3). We had a 7%", + "type": "text" + }, + { + "rank": 3, + "score": 9.08335271425718, + "content": "ic tons of CO2e in 2023—equivalent to the\",\n,to develop and scale innovations that reduce carbon,,\n,,,baseline understanding of carbon emissions from industrial\nsystems in new stores in North America starting in 2025.,,\"carbon emissions generated from driving over 11,100 cars in\",\n,\"emissions associated with the construction, operation, and\",,\n,,,\"equipment, paving the way to ultimately identifyin", + "type": "table" + }, + { + "rank": 4, + "score": 8.707361573036623, + "content": "uts us in a unique,,,\n,such as nuclear.,,\nposition to be a leader in decarbonization strategies. We,,,\nhave an opportunity to demonstrate how achieving net-,• We engage with suppliers to help reduce emissions from,,\n\"zero carbon emissions is possible across many sectors, while\",activities beyond our direct operations. We encourage,,\ncreating solutions that benefit our business as well as the,\"the", + "type": "table" + }, + { + "rank": 5, + "score": 8.18430570245479, + "content": "hanical, electrical, \nand plumbing equipment. Using data that is new to the \nindustry, Technical Memorandum 65.3 will enable Amazon, \nour supply chain partners, and our peers to establish a \nbaselin e understanding of carbon emissions from industrial \nequipment, paving the way to ultimately identifying \nadditional carbon reduction opportunities.\nServers and Hardware\nAs the world’s most comprehensi", + "type": "text" + }, + { + "rank": 6, + "score": 7.96014875338234, + "content": "ur goal to reach net-zero carbon emissions,\"these alternatives where possible, based on a number of\",,\n,,\"vehicle deployment and infrastructure, advance the\",\n,,,\"our business, we are also investing in carbon neutralization\"\n\"by 2040, 10 years ahead of the Paris Agreement—and have\",\"factors including cost, emissions reduction potential, and\",,\n,,\"deployment of carbon-free energy, modernize the gri", + "type": "table" + }, + { + "rank": 7, + "score": 7.9320909100196095, + "content": "ig goals and work backward to achieve,\n,\"many business units and subsidiaries including AWS, Devices,\"\n\"them, such as The Climate Pledge, our goal to reach net-\",\n,\"Fresh, Whole Foods Market, Amazon Private Brands, Twitch,\"\n\"zero carbon emissions by 2040, 10 years ahead of the\",\n,\"MGM Studios, and Ring.\"\nParis Agreement. We apply that same tenacity to how we,\naddress some of the world’s biggest en", + "type": "table" + }, + { + "rank": 8, + "score": 7.890755003738783, + "content": "uting chip \nefficiency, adding Low Power Mode to devices, and \ninstalling energy-efficient lighting and HVAC solutions \nin buildings. • We select lower-carbon alternatives, such as lower-\ncarbon concrete and steel in construction, and lower-\nemission fuels and vehicles in transportation. We use \nthese alternatives where possible, based on a number of \nfactors including cost, emissions reduction p", + "type": "text" + }, + { + "rank": 9, + "score": 7.868317211700303, + "content": "to set goals to decarbonize their own operations, \nand working with suppliers on initiatives to reduce their \ngreenhouse gas (GHG) emissions. We use broad carbon-free energy options to support our \ncontinued growth, enabling us to deploy and grow new \ntechnologies such as artificial intelligence (AI). By scaling \ncarbon-free energy, we aim to make Amazon a more resilient \nand more sustainable busi", + "type": "text" + }, + { + "rank": 10, + "score": 7.861685173369221, + "content": "to help reduce emissions from \nactivities beyond our direct operations. We encourage \nthem to set credible decarbonization goals, publicly share \nprogress, and implement carbon reduction strategies \nthroughout their operations and supply chains —and \nwe are providing support to help our supply chain \ntake action.1 \nIn addition to decarbonizing our own business, we are helping \ndrive progress acros", + "type": "text" + }, + { + "rank": 11, + "score": 7.661873940434669, + "content": "0,1,2\nAddressing Health Equity,,\n,,\nAWS is harnessing the power of the cloud to advance health,Funding Nature in Our Communities,\nequity globally. Through the AWS Health Equity Initiative,,\n,\"We use nature-based solutions \n to mitigate carbon emissions outside of our value\",\n\"(HEI), AWS has pledged to provide up to $60 million in\",,\n,chain and supplement the carbon reduction efforts we’re driving ", + "type": "table" + }, + { + "rank": 12, + "score": 7.623157662949613, + "content": "tions Guiding Principles Reporting Framework \n(UNGPRF). \nLearn more in our 2023 Sustainability Reporting \nFramework Summary  How to Navigate This Report\nLook for these symbols throughout the report:\n A link that directs you to a website\n A link within the report\n A link to a downloadIntroductionAbout Amazon\nAmazon is a global company with approximately \n1.5 million full- and part-time employees ", + "type": "text" + }, + { + "rank": 13, + "score": 7.6049343072448545, + "content": "le Supply Chain Human Rights Community Impact Supplier DiversityLooking Forward\nAs we move forward, Amazon aims to continue supporting \nthe communities where we operate, focusing on the areas \nwhere we can make the biggest change: education, food \nsecurity, disaster response, affordable housing equity, and \nhealth equity. Our goal is to keep using our infrastructure, \ntechnology, and passion for i", + "type": "text" + }, + { + "rank": 14, + "score": 7.4998172150174, + "content": "ustainability goals Our carbon neutralization approach focuses on three \nactions outside of our value chain. Based on climate science, \nwe know that these areas have a significant unmet need for \ninvestment and can deliver critical mitigation benefits: \n1. Reducing deforestation\n2. Advancing the removal of carbon from the atmosphere \nwith nature-based solutions\n3. Scaling up carbon removal technol", + "type": "text" + }, + { + "rank": 15, + "score": 7.466290500718765, + "content": "2015 through 2022, but with the inclusion of the additional \nprograms, actual savings was more than 3 million metric tons in 2022 \nand more than 4 million in 2023.\n 5 BloombergNEF.\n 6 Renewable natural gas (RNG) is created by decomposing organic waste \nmaterials anaerobically (without oxygen).\n 7 Electric vehicles include vans, four-wheel vehicles, three-wheel vehicles, \ntwo-wheel e-bikes, and e-m", + "type": "text" + }, + { + "rank": 16, + "score": 6.878379361978933, + "content": "(RNG).6\",,\n,,,Last Mile Electric Delivery Vehicles\n,,\"(SFC) Exchange Network, a nonprofit organization whose\",\n,\"In 2023, we piloted the use of hydrogen fuel cell vehicles\",mission is to accelerate the reduction of logistics emissions,Increasing the number of EVs in Amazon’s delivery fleet is an\n,(FCVs) in Europe and Japan. These initiatives will provide us,\"by fostering collaboration. As part of ", + "type": "table" + }, + { + "rank": 17, + "score": 6.861801886743519, + "content": "emissions by 2040, 10 years ahead of the Paris \nAgreement. We are continually working to reduce \nemissions throughout our business, as well \nas partnering across our supply chain and the \nindustries in which we operate to share and scale \nwhat we’ve learned.Goal\nInspire and empower others to sign The Climate Pledge \nand join us on a mission to reach net-zero carbon \nemissions by 2040\n473\nSignatori", + "type": "text" + }, + { + "rank": 18, + "score": 6.77550292060147, + "content": "leted and operational. We aim to reduce \nembodied carbon in building construction by using lower-\nemission concrete, lower-emission steel, and mass timber. In \n2023, 29 Amazon building projects were constructed with \nlower-carbon concrete and steel, and collectively reduced \nembodied carbon by 79,500 metric tons of CO2e, equivalent \nto the emissions generated by 17,200 cars driven for a year.\nBeca", + "type": "text" + }, + { + "rank": 19, + "score": 6.726239559833433, + "content": "taking action to achieve our,,,\"wind farm in Mississippi, and becoming the first corporate\"\n,\"initiatives, and public policy advocacy to advance access\",,\ngoals in the following ways:,,,", + "type": "table" + }, + { + "rank": 20, + "score": 6.473402333685977, + "content": "r total carbon footprint. This decrease resulted from \nreductions related to building construction, leased buildings \nand equipment, and third-party transportation, as more \ngoods were shipped by Amazon’s own logistics providers \nversus third-party providers than in 2022.\nBuilding construction is a significant driver of carbon \nemissions in many supply chains due to the associated \nembodied carbon", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/chen2024_chunks_ranked.json b/wattbot2025/data/ranked/chen2024_chunks_ranked.json new file mode 100644 index 00000000..a8d76fe7 --- /dev/null +++ b/wattbot2025/data/ranked/chen2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 3.367358090759022, + "content": "-\"\n\"the automated resource utilization overlapping optimization,\",trip time is primarily determined by network latency. In this\nwhich will be further profiled in subsection 6.4.,\"case, FHBN achieves an end-to-end latency of 33.0 µs, rep-\"\n,resenting a 50.5% reduction compared to NCCL’s 66.6 µs\n,latency. This improvement is attributed to the removal of host\n\"6.3\nNetwork Stack Optimizations\",\n,\"CPU ", + "type": "table" + }, + { + "rank": 2, + "score": 3.3235553154734467, + "content": "ive and error-prone but\nalso significantly increases maintenance complexity. Hence,\nautomated tools to help slice the models and perform relevant\noptimizations are highly desirable.\nDifficult execution overlapping. In a heterogeneous disag-\ngregated system, various devices such as compute-optimized\nGPUs, memory-optimized GPUs, and NICs can be utilized\nsimultaneously. Hence, we might achieve signif", + "type": "text" + }, + { + "rank": 3, + "score": 3.239281811178731, + "content": "lapping enabled and disabled.\nAs illustrated in Figure 14, the LLaMA-65B model experi-\nences a significant improvement in performance, achieving up\nto a 13.2% with through automated resource utilization over-\nlapping. The speedup is particularly notable for larger batch\nsizes, which produce larger KV tensors and result in greater\nlatency reduction. The effectiveness is less pronounced for the\nLLaM", + "type": "text" + }, + { + "rank": 4, + "score": 3.239281811178731, + "content": "tion overlapping.\nIn a heterogeneous disag-\",\n,5. The remote CPU waits for the RDMA receive operation\n\"gregated system, various devices such as compute-optimized\",\n,to complete.\n\"GPUs, memory-optimized GPUs, and NICs can be utilized\",\n\"simultaneously. Hence, we might achieve significant execu-\",6. The remote CPU launches the subsequent GPU kernels.\ntion time reduction if the execution of operation", + "type": "table" + }, + { + "rank": 5, + "score": 3.2121324055593115, + "content": "d-trip time from\nthe initiator GPU’s perspective, which encompasses the time\ninterval from the completion of the kernel that generates the\ndata for transmission to the start of the kernel that consumesthe received data.\nGloo NCCL (wo/ GDR) NCCL FHBN\n102104106108\nPayload size (byte)102103Round-trip time (µs)\n(a) Round-trip time.\n102104106108\nPayload size (byte)02040Bandwidth (GB/s) (b) Bandwidth ut", + "type": "text" + }, + { + "rank": 6, + "score": 3.1987276511216676, + "content": "resenting a 50.5% reduction compared to NCCL’s 66.6 µs\nlatency. This improvement is attributed to the removal of host\nCPU involvement in data transmission, eliminating expensive\nhost-device synchronization and PCIe transactions. This im-\nprovement justifies the efficacy of our fully host-bypassed\nnetwork stack design.\nFor larger payload sizes, the primary factor influencing\nnetworking time is the ", + "type": "text" + }, + { + "rank": 7, + "score": 0.0, + "content": "Efficient Heterogeneous Large Language Model Decoding\nwith Model-Attention Disaggregation\nShaoyuan Chen1Wencong Xiao2Yutong Lin1Mingxing Zhang1Yingdi Shan1Jinlei Jiang1\nKang Chen1Yongwei Wu1\n1Tsinghua University\n2ByteDance\nAbstract\nTransformer-based large language models (LLMs) exhibit\nimpressive performance in generative tasks but also intro-\nduce significant challenges in real-world serving due ", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "s in the\ntransformer-based LLMs. Specifically, the attention operator\nis memory-intensive, exhibiting a memory access pattern that\nclashes with the strengths of modern accelerators, especially\nfor long context requests.\nTo enhance the efficiency of LLM decoding, we introduce\nmodel-attention disaggregation. This approach leverages a\ncollection of cheap, memory-optimized devices for the atten-\ntion ", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "litting the attention computation over multiple devices.\nAlso, the communication bandwidth required between het-\nerogeneous devices proves to be manageable with prevalent\nnetworking technologies. To further validate our theory, we\ndevelop and deploy Lamina, an LLM inference system that\nincorporates model-attention disaggregation in a distributed\nheterogeneous cluster. Experimental results indicate", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "ce requests. The\ncore concept of disaggregation involves allocating separateresources for different tasks to improve resource utilization.\nThis approach aligns perfectly with LLM processing, which\ncan be divided into two distinct phases. The first phase, known\nas the prefill phase, processes all input tokens from the prompt\nin parallel and is computation-bound. The second phase, i.e.,\nthe decode p", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "ach phase, several methods pro-\npose using heterogeneous hardware to reduce the cost of dis-\naggregated serving [12, 59]. Specifically, flagship all-rounder\nGPUs like NVIDIA H100 integrate high-performance com-\nputational units and high-bandwidth memory (HBM) within\na single package, delivering good performance for LLM infer-\nence. However, as shown in Table 1, specialized accelerators\noptimized f", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "nd high-bandwidth internal buses within a single\nchip. Such integration leads to larger die sizes and increased\ntransistor counts, posing additional challenges for chip de-\nsigning, packaging, and thermal management [21, 25, 55], all\nof which drive up the design and manufacturing cost.\nAccording to our analyses and experiments, while the sep-\naration of resources works well for the prefill nodes, ", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "gregated KV cache for large batches, as well as the low arith-\nmetic intensity of the attention operators.\nA detailed examination reveals that the decoding phase\nmainly comprises two types of operators, each facing dis-\ntinct resource bottlenecks. Linear transformations, includ-\n1arXiv:2405.01814v2 [cs.LG] 10 Apr 2025\n\nTable 1: H100, H20, and TPU v6e specifications.\nH100 H20 TPU v6e [7]\nBF16 TFL", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "ailable on cloud service providers, the listed price\nis estimated using the relative complete system cost against H100.\ning QKVO projections and feedforward networks, are im-\nplemented with generalized matrix-matrix multiplications\n(GEMMs). Since all requests multiply with the same parame-\nter matrices in these operators, processing multiple requests in\nbatch can avoid repeated parameter loads fro", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "h sizes does\nnot improve the computation resource utilization but places\nadditional pressure on the already limited memory capacity.\n1.2 Our Contributions\nIn light of the above findings, we propose an innovative con-\ncept called model-attention disaggregation , as illustrated\nin Figure 1. This approach involves further disaggregating\nthe decoding phase by creating two pools of heterogeneous\naccele", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "ind of operators, this architecture further increases\nhardware utilization and leads to better overall performance.\nMoreover, different LLMs and workloads present varying\ncomputation and memory resource requirements. Homoge-\nneous accelerator solutions, however, can only provide a fixed\nratio of computation and memory resources , which can\nresult in resource wastage. For instance, as context lengt", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "ach\nkind of accelerators to better match the LLM and workload\nand hence improve resource utilization.\nThe primary challenge associated with attention offload-\ning arises from the substantial communication demands be-\nModel/Attention Disaggregation\nKV Cache ①\n②Prefill/Decode\nDisaggreagtion\nModel\nWeights\nCompute-\nOptimized GPUKV Cache\nMemory-\nOptimized GPU\nPagedCache\nManagerContinuous\nBatchingReques", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "tors. Unlike the\noriginal prefill-decode disaggregation, where the KV cache\nis transferred only once between the prefill nodes and the\ndecode nodes, our model-attention disaggregation architec-\nture requires inter-GPU communication for every layer of the\nmodel. Even worse, communication between heterogeneous\nGPUs must rely on data center networks (DCNs), such as\nEthernet and InfiniBand, which prov", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "y of our novel disaggregated ar-\nchitecture, we first conduct a detailed quantitative study in-\ndicating that these concerns are manageable in the context\nof LLM inference. In subsection 3.1, we provide profiling\nand analysis to determine the minimum bandwidth threshold\nbetween different accelerator pools. Our findings reveal that\n200/400Gbps DCNs, widely deployed in current AI-oriented\ndata cente", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "two specific techniques to reduce the network-\ning overhead. First, we designed and deployed a fully host-\nbypassed network stack. Leveraging PCIe P2P capabilities,\nthis revamped network stack enables GPUs to directly talk\nwith network interface cards (NICs), eliminating the need\nfor host CPU synchronization and involvement for network\ntransmissions. The network data is also directly read from\nand", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/chung2025_chunks_ranked.json b/wattbot2025/data/ranked/chung2025_chunks_ranked.json new file mode 100644 index 00000000..d95c1183 --- /dev/null +++ b/wattbot2025/data/ranked/chung2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 8.270308979716624, + "content": "s): Average power draw can be calculated by dividing total energy,,,,,,,,,,,,\n,consumption by the time taken to run the benchmark. This is useful for power provisioning.,,,,,,,,,,,,\n,• Monetary cost ($): Monetary cost can be calculated by multiplying energy consumption by the,,,,,,,,,,,,\n,cost of electricity in the region and time frame in which the benchmark was run.,,,,,,,,,,,,\n,• Operational ca", + "type": "table" + }, + { + "rank": 2, + "score": 7.382568397923224, + "content": "ars behind each solid bar are estimations based on the GPU’s TDP, with numbers showing\nthe ratio of overestimation. Note the log scale Y-axis.\nMetrics. Energy consumption (Joules) reported by the ML.ENERGY Benchmark is a fundamental\nquantity that can be used to derive other useful metrics.\n•Average power draw (Watts) : Average power draw can be calculated by dividing total energy\nconsumption by th", + "type": "text" + }, + { + "rank": 3, + "score": 6.218229076649438, + "content": "ohan Yan, Hasan\nGenc, Grace Dinh, Qijing Huang, Kurt Keutzer, Michael W. Mahoney, Sophia Shao, and Amir\nGholami. Full stack optimization of transformer inference. Architecture and System Support\nfor Transformer Models , 2023.\n[34] Woosuk Kwon, Zhuohan Li, Siyuan Zhuang, Ying Sheng, Lianmin Zheng, Cody Hao Yu,\nJoseph Gonzalez, Hao Zhang, and Ion Stoica. Efficient memory management for large languag", + "type": "text" + }, + { + "rank": 4, + "score": 6.016895361492303, + "content": "ty estimates the greenhouse gas emissions\nassociated with the electricity consumed. It can be calculated by multiplying energy consumption\nby the carbon intensity of the particular region and time frame in which the benchmark was run.\n4 Results Highlight\nIn this section, we highlight notable results from the ML.ENERGY Benchmark; the full set of\nresults is available on the ML.ENERGY Leaderboard.3Th", + "type": "text" + }, + { + "rank": 5, + "score": 3.803028321050168, + "content": "apacity of a single GPU. This requires\nmultiple GPUs to execute inference for a single model, and GPUs must constantly communicate with\neach other to do so [58].\nIn order to ablate the effect of communication, we employ the same Llama 3.1 8B model and vary\nthe number of GPUs used (Figure 9). Because the amount of computation executed is the same\nregardless of the number of GPUs, energy consumption", + "type": "text" + }, + { + "rank": 6, + "score": 3.492094166077999, + "content": "ween the GPUs offsets the\nreduction in computation time. Since communication time increases with the number of GPUs, using\ntoo many GPUs can lead to slowdowns in executing the same amount of computation and increase\nenergy consumption.\n15\n\n0 200 400 600 800 1000\nBatch size0200400600Power draw (W)\nA100 TDP (max power draw)H100 TDP (max power draw)\nA100 H100(a) Llama 3.1 8B\n0 200 400 600 800 1000\nBa", + "type": "text" + }, + { + "rank": 7, + "score": 3.386991346470018, + "content": "r code generated by\nchatGPT really correct? rigorous evaluation of large language models for code generation. In\nNeurIPS , 2023.\n[41] Anton Lozhkov, Raymond Li, Loubna Ben Allal, Federico Cassano, Joel Lamy-Poirier, Noua-\nmane Tazi, Ao Tang, Dmytro Pykhtar, Jiawei Liu, Yuxiang Wei, et al. Starcoder 2 and the stack\nv2: The next generation. arXiv preprint arXiv:2402.19173 , 2024.\n[42] Alexandra Sash", + "type": "text" + }, + { + "rank": 8, + "score": 3.1209486650314147, + "content": "s in the cloud.\n\"ASPLOS, 2024.\"\n\"[51] David Patterson,\nJoseph Gonzalez, Quoc Le, Chen Liang, Lluis-Miquel Munguia, Daniel\"\n\"Rothchild, David So, Maud Texier, and Jeff Dean. Carbon emissions and large neural network\"\n\"training. arXiv preprint, 2021.\"\n\"[52] Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas Müller,\"\n\"Joe Penna, and Robin Rombach. SDXL: Improving latent d", + "type": "table" + }, + { + "rank": 9, + "score": 3.1075587181115893, + "content": "Sora. https://openai.com/index/sora , 2024.\n[50] Pratyush Patel, Esha Choukse, Chaojie Zhang, Íñigo Goiri, Brijesh Warrier, Nithish Mahalingam,\nand Ricardo Bianchini. Characterizing power management opportunities for llms in the cloud.\nASPLOS , 2024.\n[51] David Patterson, Joseph Gonzalez, Quoc Le, Chen Liang, Lluis-Miquel Munguia, Daniel\nRothchild, David So, Maud Texier, and Jeff Dean. Carbon emis", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "arXiv:2505.06371v1 [cs.LG] 9 May 2025The ML.ENERGY Benchmark: Toward Automated\nInference Energy Measurement and Optimization\nJae-Won Chung Jiachen Liu Jeff J. Ma Ruofan Wu\nOh Jun Kweon Yuxuan Xia Zhiyu Wu Mosharaf Chowdhury\nUniversity of Michigan\nAbstract\nAs the adoption of Generative AI in real-world services grow explosively, energy\nhas emerged as a critical bottleneck resource. However, energ", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "derboard, which have\nserved as a valuable resource for those hoping to understand and optimize the\nenergy consumption of their generative AI services. In this paper, we explain four\nkey design principles for benchmarking ML energy we have acquired over time,\nand then describe how they are implemented in the ML.ENERGY Benchmark.\nWe then highlight results from the latest iteration of the benchmark, ", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "g computed by the\nmodel. The ML.ENERGY Benchmark is open-source and can be easily extended\nto various customized models and application scenarios.\n1 Introduction\nGenerative AI models have rapidly transitioned from research prototypes to real-world services such\nas ChatGPT [48], Character AI [4], Sora [49], and Midjourney [45]. However, exponential growth\nrarely continues without facing scaling bot", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "ow, and sometimes impossible. This particularly impacts\nserving real-world services as ML inference reportedly accounts for 80–90% of the total compute\ndemand [10, 27, 50, 51]. Left unaddressed, the energy bottleneck will not only hinder AI research and\ndevelopment progress [26], but also lead to energy being either squeezed out of existing electricity\ngrids and impacting availability and price [3", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "nergy measurement and accounting during\nexecution, let alone optimization? To bridge this gap, we launched the ML.ENERGY Leaderboard,1\nthe first inference energy leaderboard for modern generative AI models to the best of our knowledge.\nThe Leaderboard has been gradually expanding in multiple dimensions to now include (1) 40 different\ngenerative AI model architectures across a wide range of tasks –", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "U type = A100……Max batch size = 256Pipeline parallel = 1Tensor parallel = 2GPU type = H100123\nTime (s)Energy (J)\nTarget4\nMeasured energy & latencyFigure 1: Overview of the benchmarking and optimization flow of the ML.ENERGY Benchmark.\ncoding, VLM [38] visual chat, and text-to-image, text-to-video, and image-to-video generation using\nDiffusion models [12, 20] – and (2) more up-to-date hardware and ", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "provides an easily extensible benchmark suite and a comprehensive\nset of tools for measuring the inference energy consumption of generative AI models for various\ntasks under realistic deployment environments.\n•Automated Optimization : Based on energy measurement results, it provides automated energy\noptimization recommendations for generative AI model deployment.\nFinally, we highlight notable resu", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "cked by automated optimization (Section 4).\nThe ML.ENERGY Benchmark is open-source on GitHub,2and the ML.ENERGY Leaderboard allows\neveryone to browse full results from the ML.ENERGY Benchmark. The benchmark is designed to be\neasily extensible, allowing users to benchmark their models and compare them to others without the\nheavy lifting of building their own benchmarking dataset, runtime, and analy", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": ", and ultimately actionable.\n2.1 Generalizability and Portability\nGoal. Every computer system is configured with different hardware and software components, and\nmeasurements from a particular system will never truly represent those from another system. For\ninstance, systems can be configured with different CPU and DRAM models, and running different\n2https://github.com/ml-energy/leaderboard\n2\n\nLinu", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "provide generalizable insights and recommendations across a wide range of systems.\nOur approach. We focus on software-based GPU energy measurement for the following reasons:\n•GPUs are the dominant worker and energy consumer in a system running ML services, accounting\nfor 50–70% of the total provisioned power in the datacenter [50].\n•Compared to other hardware components, GPU models are more standa", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "senting Real-World Deployments\nGoal. Benchmarking results often inform real-world deployment optimizations, are used to plan\nfuture energy usage, affect the design of new hardware and software systems, and serve as base num-\nbers for long term projections that affect policymaking. Therefore, it is crucial that our measurements\nrepresent those from real-world deployments as closely as possible.\nOur", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/cottier2024_chunks_ranked.json b/wattbot2025/data/ranked/cottier2024_chunks_ranked.json new file mode 100644 index 00000000..77db2ee6 --- /dev/null +++ b/wattbot2025/data/ranked/cottier2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 0.0, + "content": "THE RISING COSTS OF TRAINING FRONTIER AIMODELS\nBen Cottier1Robi Rahman1,2\nLoredana Fattorini2Nestor Maslej2Tamay Besiroglu1David Owen1\nABSTRACT\nThe costs of training frontier AI models have grown dramatically in recent years, but there is limited\npublic data on the magnitude and growth of these expenses. This paper develops a detailed cost\nmodel to address this gap, estimating training costs using", + "type": "text" + }, + { + "rank": 2, + "score": 0.0, + "content": "Gemini, the most significant expenses\nare AI accelerator chips and staff costs, each costing tens of millions of dollars. Other notable costs\ninclude server components (15-22%), cluster-level interconnect (9-13%), and energy consumption\n(2-6%). If the trend of growing development costs continues, the largest training runs will cost more\nthan a billion dollars by 2027, meaning that only the most we", + "type": "text" + }, + { + "rank": 3, + "score": 0.0, + "content": "economic\nanalysis [ 2] and the discovery of empirical scaling laws, which show that model performance improves with more\nparameters and training data [ 3,4]. Dario Amodei, CEO of the AI lab Anthropic, has stated that frontier AI developers\nare likely to spend close to a billion dollars on a single training run this year, and up to ten billion-dollar training runs\nin the next two years [ 5]. Given ", + "type": "text" + }, + { + "rank": 4, + "score": 0.0, + "content": "the public domain. In collaboration with Epoch AI, the 2024 AI Index presented one of\nthe most comprehensive datasets to date, estimating the costs of training runs based on cloud rental prices [ 6]. We\nbuild on that work with a more in-depth account of hardware, energy and R&D staff costs for both training runs and\nexperiments, as well as a more detailed analysis of how costs are increasing over ", + "type": "text" + }, + { + "rank": 5, + "score": 0.0, + "content": "he cost of frontier\nmodels. The first approach estimates the hardware capital expenses (CapEx) amortized over the final training run,\nalong with the cost of hardware energy consumption. By considering AI accelerator chips, other server hardware,\nnetworking hardware, and energy separately, this approach can provide more accurate training costs. We find that the\nmost expensive publicly-announced tra", + "type": "text" + }, + { + "rank": 6, + "score": 0.0, + "content": "are this approach to the cloud-price approach that was first presented in the AI Index [ 6]. Instead of\nestimating hourly compute costs in detail, the cloud-price approach simply uses historical rental rates from cloud\nplatforms. The cloud-price approach shows a similar growth rate ( 2.5×per year with a 90% CI of 2.1×to3.1×), but\n1Epoch AI.2Stanford University.arXiv:2405.21015v2 [cs.CY] 7 Feb 20", + "type": "text" + }, + { + "rank": 7, + "score": 0.0, + "content": "ed for cluster-level networking. Open circles\nindicate costs which used an estimated production cost of Google TPU hardware. These costs are generally more\nuncertain than the others, which used actual price data rather than estimates.\nyields costs that are about twice as large on average. We expect the cloud-price approach to overestimate frontier model\ncosts, since model developers usually either", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "velopment\nof the model (i.e. both experiments and training). We select four especially notable models for this approach—GPT-3,\nOPT-175B, GPT-4, and Gemini Ultra. For these models, we find that R&D staff costs including equity are between\n29% and 49% of the total amortized cost. Computing hardware makes up 47–64%, while energy comprises only 2–6%.\nHowever, if we exclude equity the fraction for R&D ", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "s sheds light on not only current costs but also the\neconomic hurdles that lie ahead as AI continues to scale.\nAll of our results can be reproduced using the code and data available at https://github.com/epoch-research/\ntraining-cost-trends .\n2 Methodology\n2.1 Datasets and frontier model selection\nOur investigation draws upon the Notable AI Models database, which documents 796 notable models acros", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "r 2015 (the start of the large-scale ML era according to [ 8])\nand up to 31 December 2023. This resulted in 276 selected models. For these models, we recorded the training time,\nhardware type and quantity, and utilization rate sourced from each model’s original publication, where possible.\nFor our main results, we examined 41 models that were historically at the frontier of compute. Specifically, ", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "ting costs.\n2\n\nIn addition to the data on machine learning models, we compiled a dataset of historical hardware prices, allowing us to\nestimate training costs. This price dataset contained cloud rental prices and hardware purchase prices for 24 different\nhardware models (e.g. NVIDIA A100) between 2015 and 2023. In total there were 142 entries, 52 of which were\npurchase prices and 90 of which were ", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "timating\nproduction costs for TPUs. Further details are provided in Appendix A.1.\nHardware normally remains available for future use after a training run finishes, but its value depreciates over time due\nto hardware progress. We amortized the cost of a training run based on this depreciation. Specifically, we depreciated\nthe value of hardware at a rate of r= 0.14orders of magnitude per year, based", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "years. For example, if the training run starts one year after hardware is\nacquired, the start value is approximately 72% of the acquisition cost. We neglected the impact of hardware failures on\ndepreciation, as the effect seemed small compared to hardware progress. We provide evidence for that in Appendix A.3.\nAfter finding the initial value of the hardware, the amortized cost of the training run ", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "ed chip-hours for the training time and the number of chips, using a linear approximation. This\nled to our final formula for amortized training cost:\nAmortized training cost ≈Start value per chip ×Training chip-hours\n(365×24)hours/year×rln 10\nUp until Section 3.5, our results only account for the chip-hours of the final training run. In Section 3.5, we scale up the\nchip-hours to account for all ex", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "gy consumption cost\nIn addition to the capital costs of hardware, we also considered the cost of energy consumed by hardware during model\ntraining. We estimated this using the following formula:\nTotal energy cost of training =Energy cost rate ($/kWh) ×Hardware TDP (kW) ×\nAverage power to TDP ratio (%) ×Data center PUE ×Training chip-hours (h)\nwhere TDP is thermal design power and PUE is power usag", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "d on hardware manufacturers’ literature. However, some parameters such as average power to TDP\nratio could not be found in technical specifications and had to be estimated. For references and method details, see\nAppendix A.4.\n2.4 Cloud compute cost\nWhile the amortized hardware CapEx + energy approach is a bottom-up method that accounts for hardware and energy\ncosts, cloud rental prices offer a sim", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "ion and tensor number\nformats would make the rate faster, but this was not estimated. This also assumes that hardware improves continuously. In reality,\nhardware improves in increments with each new release.\n3\n\nwith those derived from cloud rental prices, we can validate our approach and provide a more comprehensive picture\nof AI training costs. The cloud approach also allows estimates of model tr", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "loud rental prices, we used the following formula:\nTotal cost =Price per chip-hour ×Training chip-hours\nThe price per chip-hour was obtained from our hardware price database, which includes prices for various hardware\ntypes, cloud providers, and rental dates. We matched the hardware type and publication date of each ML model with the\nmost appropriate price, using the developer of the ML model to d", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "elop-\nment surrounding it is crucial. We therefore used a third approach that considers all of the compute that went into model\ndevelopment, as well as the cost of R&D staff developing the model. Since this approach was more time-intensive, and\nrelied on having a list of contributors to estimate R&D staff cost, we applied it to just four models: GPT-3, OPT-175B,\nGPT-4, and Gemini Ultra.\nTo estimat", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "infrastructure at Meta.\nAppendix A.6 provides further details. Based on this, we sampled the factor from a log-normal distribution with a 90%\nCI of 1.2x to 4x, meaning that total compute for model development is 1.2x to 4x larger than the final training run.\n2.5.1 R&D staff costs\nResearch and development (R&D) staff costs are an often-neglected component of the total cost of developing ML\nmodels. ", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/dodge2022_chunks_ranked.json b/wattbot2025/data/ranked/dodge2022_chunks_ranked.json new file mode 100644 index 00000000..47c9e96f --- /dev/null +++ b/wattbot2025/data/ranked/dodge2022_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 6.067754851013553, + "content": "cifically, the SCI uses a \"consequential\" carbon accounting approach, which aims to quantify the marginal change in\nemissions caused by decisions or interventions. This differs from the commonly used \"attributional\" carbon accounting\napproach, which uses average carbon intensity data, meaning it does not provide the most actionable information to\nhelp reduce carbon emissions. Due to the myriad pot", + "type": "text" + }, + { + "rank": 2, + "score": 5.662717558775276, + "content": "enter plays a significant role in the carbon intensity for\n\"a given cloud instance, and find that choosing an appropriate region can have the largest operational emissions reduction impact. We\"\n\"also present new results showing that the time of day has meaningful impact on operational software carbon intensity.Finally, we\"\nconclude with recommendations for how machine learning practitioners can us", + "type": "table" + }, + { + "rank": 3, + "score": 5.602384503884998, + "content": "carbon intensity for\na given cloud instance, and find that choosing an appropriate region can have the largest operational emissions reduction impact. We\nalso present new results showing that the time of day has meaningful impact on operational software carbon intensity.Finally, we\nconclude with recommendations for how machine learning practitioners can use software carbon intensity information to", + "type": "text" + }, + { + "rank": 4, + "score": 5.539398340826914, + "content": "ch aims to quantify the marginal change in\"\n\"emissions caused by decisions or interventions. This differs from the commonly used \"\"attributional\"\" carbon accounting\"\n\"approach, which uses average carbon intensity data, meaning it does not provide the most actionable information to\"\nhelp reduce carbon emissions. Due to the myriad potential pitfalls of relying on market-based measures in place of\n\"a", + "type": "table" + }, + { + "rank": 5, + "score": 5.31341952264698, + "content": "high emissions and very low emissions, and\nthus Pause and Resume can lead to significant reductions. However, other regions do not present as much variance,\nand thus lead to less reduction in emissions. See Figures 3 and 4. The lack of geographic diversity in the region list is\nan unfortunate consequence of the unavailability of carbon intensity data from other continents; we hope such data\nbecome", + "type": "text" + }, + { + "rank": 6, + "score": 4.961725706120289, + "content": "y\nWest EuropeNorth EuropeNorwayUK SouthAustralia051015202530CO2 emissions decrease in %25%\n50%\n75%\n100% (b)Pause and Resume optimization for 6B parameters Transformer.\nFig. 4. What proportion of emissions can we expect to save if we pause an AI workload when emissions in a region are high and\nresume when emissions are low, increasing the total duration by up to double the original duration? For sh", + "type": "text" + }, + { + "rank": 7, + "score": 4.182917165402106, + "content": "to about 25%. We confirmed with WattTime that emissions estimates for West US were correct, as that region has large\nvariance.\n6.1.2 Comparable Duration Increases. In the previous section we examined the amount of emissions reduction for\nour two algorithms by region, and compared Pause and Resume increasing duration by a proportion of the original\nexperiment and Flexible Start by a fixed duration.", + "type": "text" + }, + { + "rank": 8, + "score": 3.547281611519623, + "content": "than 30% in multiple regions,\nand up to 80% in West US; for very long runs like training a 6 billion parameter language model for 8 days (b), changing the start time\nby up to 24 hours leads to less than 1.5% reduction at best in any region. Note: we confirmed with WattTime that emissions estimates\nfor West US were correct, that region has large variance.\nduring those intervals and compute the corr", + "type": "text" + }, + { + "rank": 9, + "score": 3.547281611519623, + "content": "0\n\"and up to 80% in West US; for very long runs like training a 6 billion parameter language model for 8 days (b), changing the start time\"\nby up to 24 hours leads to less than 1.5% reduction at best in any region. Note: we confirmed with WattTime that emissions estimates\n\"for West US were correct, that region has large variance.\"\nduring those intervals and compute the corresponding emissions. We ", + "type": "table" + }, + { + "rank": 10, + "score": 3.270534195996002, + "content": "on of this paper is the simplest: a,\npresentation of the software carbon intensity (SCI) as a proxy for carbon emissions for a given cloud instance as it is,\nrunning.,\n\"3.1\nMethodology: Computing CO2 Intensity\",\n\"In this section we describe a method for estimating carbon intensity for cloud instances. At a high level, this involves\",\n\"tracking electricity consumption of hardware related to a singl", + "type": "table" + }, + { + "rank": 11, + "score": 3.090977841120739, + "content": "tensity measurement ( 𝐼). Once more this can be further refined to simply:\n𝑆𝐶𝐼=𝐶per𝑅 (3)\n4\n\nMeasuring the Carbon Intensity of AI in Cloud Instances FAccT ’22, June 21–24, 2022, Seoul, Republic of Korea\nwhere𝐶=𝑂+𝑀is the software carbon intensity for a given cloud instance. In this paper, we focus on measuring\noperational emissions 𝑂, and leave measurement and accounting for embodied emissions due t", + "type": "text" + }, + { + "rank": 12, + "score": 3.0866916391297647, + "content": "icity consumption of hardware related to a single cloud instance, and mapping that electricity usage to\nCO2emissions by using a grid-based carbon intensity.\nAs developed by the Green Software Foundation, the Software Carbon Intensity ( 𝑆𝐶𝐼) is a rate, carbon emissions per\none functional unit, or R. The equation used to calculate the 𝑆𝐶𝐼value of a software system is therefore:\n𝑆𝐶𝐼=((𝐸∗𝐼)+𝑀)per𝑅 (1)", + "type": "text" + }, + { + "rank": 13, + "score": 3.0834868048108928, + "content": "0\n\"Measuring the Carbon Intensity of AI in Cloud Instances\nFAccT ’22, June 21–24, 2022, Seoul, Republic of Korea\"\n\"where 𝐶 = 𝑂 + 𝑀 is the software carbon intensity for a given cloud instance. In this paper, we focus on measuring\"\n\"operational emissions 𝑂, and leave measurement and accounting for embodied emissions due to specialized ML hardware\"\nsuch as GPUs to future work (see §8).\nThe objective ", + "type": "table" + }, + { + "rank": 14, + "score": 3.07192430247256, + "content": ", the computational demands of which incur a high energy cost and a\ncommensurate carbon footprint. As a result, recent scholarship has called for better estimates of the greenhouse gas impact of AI: data\nscientists today do not have easy or reliable access to measurements of this information, which precludes development of actionable\ntactics. We argue that cloud providers presenting information ab", + "type": "text" + }, + { + "rank": 15, + "score": 2.998958101248194, + "content": "uch as machine learning, the computational demands of which incur a high energy cost and a\"\n\"commensurate carbon footprint. As a result, recent scholarship has called for better estimates of the greenhouse gas impact of AI: data\"\n\"scientists today do not have easy or reliable access to measurements of this information, which precludes development of actionable\"\ntactics. We argue that cloud provide", + "type": "table" + }, + { + "rank": 16, + "score": 2.9457213147497834, + "content": "of carbon dioxide equivalent per kilowatt-hour of electricity (gCO 2eq/kWh)\n•𝑀=Embodied carbon (also referred to as “embedded carbon”) is the amount of carbon emitted during the\ncreation, usage, and disposal of a hardware device. When software runs on a device, a fraction of the total\nembodied emissions of the device is allocated to the software.\n•𝑅=Functional unit. In this instance, we are defini", + "type": "text" + }, + { + "rank": 17, + "score": 2.916176182819811, + "content": "0,1\n,(1)\nwhere:,\n,\"• 𝐸 = Energy consumed by a software system. Specifically, we focus on energy consumption of Graphical Processing\"\n\"Units, or GPUs. The units used are kilowatt-hours (kWh).\",\n,• 𝐼 = Location-based marginal carbon emissions for the grid that powers the datacenter. WattTime provides\nmeasurements of grams of carbon dioxide equivalent per kilowatt-hour of electricity (gCO2eq/kWh),\n• ", + "type": "table" + }, + { + "rank": 18, + "score": 2.881688600970552, + "content": "ligible but beyond the scope of this paper. Finally, for workloads that do not use GPUs (e.g., storage\",,,,,,,\n,\"or web hosting) we recommend users choose low emissions regions and times of day, as they will not have access to\",,,,,,,\n,single-instance emissions calculations. We leave it open for future research to address how to appropriately allocate,,,,,,,\n,CO2 emissions from such data center-wi", + "type": "table" + }, + { + "rank": 19, + "score": 2.8655566026074037, + "content": "0\n\"Measuring the Carbon Intensity of AI in Cloud Instances\nFAccT ’22, June 21–24, 2022, Seoul, Republic of Korea\"\n\"Foundation’s guidelines regarding Software Carbon Intensity (SCI), and suggest future areas of research to improve the\"\nstate of carbon estimation and reporting in AI.\n\"2\nRELATED WORK\"\n\"Attention was first drawn to the environmental impact of AI research by the seminal work of Strubel", + "type": "table" + }, + { + "rank": 20, + "score": 2.815830515023203, + "content": "fic marginal emissions data per energy unit. We provide measurements of operational\n\"software carbon intensity for a set of modern models covering natural language processing and computer vision applications, and a\"\n\"wide range of model sizes, including pretraining of a 6.1 billion parameter language model. We then evaluate a suite of approaches for\"\n\"reducing emissions on the Microsoft Azure clou", + "type": "table" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/ebert2024_chunks_ranked.json b/wattbot2025/data/ranked/ebert2024_chunks_ranked.json new file mode 100644 index 00000000..6b0f6abd --- /dev/null +++ b/wattbot2025/data/ranked/ebert2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 8.099135407757808, + "content": "hao, and Zishang Zhu. 2023. China’s green data center\"\n\"nance.\nhttps://www.finance.gov.au/sites/default/files/2023-11/Net_Zero_\",\"development:Policies and carbon reduction technology path.\nEnvironmental\"\nGovernment_Operations_Strategy.pdf,\"Research 231 (2023), 116248. doi:10.1016/j.envres.2023.116248\"\n[22] Department of General Services. 2014. Management Memo MM 14-09: Energy,\"[49]\nPengfei Li, Jia", + "type": "table" + }, + { + "rank": 2, + "score": 7.485983308615928, + "content": "sevier, 293–316.\n[48] Guozhu Li, Zixuan Sun, Qingqin Wang, Shuai Wang, Kailiang Huang, Naini Zhao,\nYanqiang Di, Xudong Zhao, and Zishang Zhu. 2023. China’s green data center\ndevelopment:Policies and carbon reduction technology path. Environmental\nResearch 231 (2023), 116248. doi:10.1016/j.envres.2023.116248\n[49] Pengfei Li, Jianyi Yang, Mohammad A Islam, and Shaolei Ren. 2023. Making ai\nless\" thir", + "type": "text" + }, + { + "rank": 3, + "score": 5.662698505035185, + "content": "271 (2023).\n\"[23]\nJesse Dodge, Taylor Prewitt, Remi Tachet des Combes, Erika Odmark, Roy\",\"[50]\nPengfei Li, Jianyi Yang, Adam Wierman, and Shaolei Ren. 2023. Towards En-\"\n\"Schwartz, Emma Strubell, Alexandra Sasha Luccioni, Noah A Smith, Nicole\",\"vironmentally Equitable AI via Geographical Load Balancing.\narXiv preprint\"\n\"DeCario, and Will Buchanan. 2022. Measuring the carbon intensity of ai in clo", + "type": "table" + }, + { + "rank": 4, + "score": 5.568619006395051, + "content": "arxiv.org/abs/2307.05494\n[51] Alexandra Sasha Luccioni and Alex Hernandez-Garcia. 2023. Counting carbon:\nA survey of factors influencing the emissions of machine learning. arXiv preprint\narXiv:2302.08476 (2023).\n[52] Alexandra Sasha Luccioni, Emma Strubell, and Kate Crawford. 2025. From\nEfficiency Gains to Rebound Effects: The Problem of Jevons’ Paradox in AI’s\nPolarized Environmental Debate. arXi", + "type": "text" + }, + { + "rank": 5, + "score": 4.396672863762926, + "content": "ansparency and risk management for high-risk AI and\n\"towards more substantive goals and obligations. Indeed, the AI Act\",\n,general-purpose AI systems make some progress in requiring doc-\ndoes contain some language to this effect. For providers of GPAI,\n,\"umentation of computational resources and energy consumption,\"\n\"models with systemic risk and providers of HRAI systems, the Act\",\n,\"but signific", + "type": "table" + }, + { + "rank": 6, + "score": 4.216009420376099, + "content": "enerally, it is\narguably crucial to move beyond mere transparency provisions\ntowards more substantive goals and obligations. Indeed, the AI Act\ndoes contain some language to this effect. For providers of GPAI\nmodels with systemic risk and providers of HRAI systems, the Act\nmandates risk assessment and mitigation (Art. 55(1)(b) and Art. 9).\nWe argue that these measures should also consider environm", + "type": "text" + }, + { + "rank": 7, + "score": 4.017424873893043, + "content": "Conference’17, July 2017, Washington, DC, USA Kai Ebert, Nicolas Alder, Ralf Herbrich, and Philipp Hacker\nB Policy Proposals Overview\nTable 1: Summary of Policy Proposals for AI and Environ-\nmental Impact\nArea Policy Proposal and Description\nEnergy and Environmental\nReportingEnergy consumption from inferences : Include energy consumption from\nboth single and cumulative inferences in reporting.\nInd", + "type": "text" + }, + { + "rank": 8, + "score": 3.1961582121565457, + "content": "netary limits [8–10, 18].\"\n\"SIA), and provide guidance on its operationalization. The EU data\",Regulatory frameworks are beginning to address these chal-\ncenter regulation proves to be a good first step but requires further,\"lenges, too. While the new U.S. administration scraps environmen-\"\ndevelopment by including binding renewable energy and efficiency,\"tal rules and generally deregulates, the E", + "type": "table" + }, + { + "rank": 9, + "score": 3.1698266231432606, + "content": "bligations; Legal and Regulatory Clarifications; Transparency and,\"as one of its core goals and contains dedicated sustainability rules,\"\nAccountability Mechanisms; and Future Far-Reaching Measures,which also apply to US and other non-EU providers offering models\nbeyond Transparency.,\"in the EU. Similarly, the Digital Services Act (DSA) compels Very\"\n,Large Online Platforms and Very Large Online S", + "type": "table" + }, + { + "rank": 10, + "score": 3.093372265975072, + "content": "der the AI Continent,the problem of sustainable AI remains underexplored. Existing\n\"Action Plan, including the establishment of AI Gigafactories [25],\",\"contributions date from before the AI Act’s final version [63], which\"\nis a late but necessary impetus for technological and strategic au-,\"differs significantly from previous proposals (see below, 5.2.) or do\"\ntonomy but must not endanger the Uni", + "type": "table" + }, + { + "rank": 11, + "score": 3.068700514376692, + "content": "-based\nsocio-technical systems, extracting and, at times, exploiting, both\nnatural and human resources [ 18,24,39,42,58], often particularly\nfrom marginalized communities [ 61,65] and regions [ 50,80]. Both\nthe quest for performance and the funding logic behind AI force\nproviders to scale them to a point where this trajectory may push\nagainst planetary limits [8–10, 18].\nRegulatory frameworks are ", + "type": "text" + }, + { + "rank": 12, + "score": 3.068700514376692, + "content": "on\nas one of its core goals and contains dedicated sustainability rules,\nwhich also apply to US and other non-EU providers offering models\nin the EU. Similarly, the Digital Services Act (DSA) compels Very\nLarge Online Platforms and Very Large Online Search Engines to\nthoroughly assess and mitigate systemic risks, which is of particu-\nlar relevance for hybrid platforms increasingly integrating AI. ", + "type": "text" + }, + { + "rank": 13, + "score": 3.044419196860386, + "content": "nvironmental\nimpact of\"\n\"risks, in keeping with the normative goals of the AI Act listed in\",\n,\"AI applications. Moreover, transparency measures are restricted to\"\n\"Article 1 and Recitals 1, 2 and 8.\",\n,\"authorities, limiting broader accountability and public scrutiny.\"\n\"Crucially, both provisions relate to risks of the AI model or sys-\",\n,\"Additionally, while the Act imposes risk assessment and m", + "type": "table" + }, + { + "rank": 14, + "score": 3.032422066041649, + "content": "shortcomings\nthreaten to undermine the most promising legislative avenues for\ntackling AI’s climate effects. The EU’s recent push for massive\ninvestments in computing infrastructure under the AI Continent\nAction Plan, including the establishment of AI Gigafactories [ 25],\nis a late but necessary impetus for technological and strategic au-\ntonomy but must not endanger the Union’s climate goals unde", + "type": "text" + }, + { + "rank": 15, + "score": 3.032422066041649, + "content": "s, the AI Act requires no such reporting specifi-\",\"and to whom, is yet to be determined.\"\ncally for AI—as it does for energy use—nor does it cover operations,\noutside the EU.,\n,\"5.5\nDiscussion and Interim Conclusion on the\"\n,AI Act\n\"5.4\nEnvironmental Risk Assessment and\",\"Overall, while the AI Act introduces valuable steps toward address-\"\n,\"ing climate-related concerns in AI development and depl", + "type": "table" + }, + { + "rank": 16, + "score": 3.0016470314665358, + "content": "oni, Sylvain Viguier, and Anne-Laure Ligozat. 2023. Esti-\"\n\"[26]\nEuropean Commission Joint Research Centre. 2023.\nEU Code of Conduct\nfor\",\"mating the carbon footprint of bloom, a 176b parameter language model. Journal\"\n\"Data Centres: Towards More Innovative, Sustainable, and Secure Data Centre\",\"of Machine Learning Research 24, 253 (2023), 1–15.\"\n\"https://joint-research-centre.ec.europa.eu/jrc-new", + "type": "table" + }, + { + "rank": 17, + "score": 2.926653759270999, + "content": "ournal of the Faculty of Information 7, 2 (2022), 5–11.\"\n\"https://techreg.org/article/view/2025-1-Commins-Irion\nRegulation 2025 (2025).\",[42] Tugrul Keskin and Ryan David Kiggins (Eds.). 2021. Towards an International\n\"[17]\nJosh Cowls, Andreas Tsamados, Mariarosaria Taddeo, and Luciano Floridi.\",Political Economy of Artificial Intelligence. Palgrave Macmillan.\n\"2023.\nThe AI gambit:\nleveraging arti", + "type": "table" + }, + { + "rank": 18, + "score": 2.914240179647611, + "content": "0,1\nTable 1: Summary of Policy Proposals for AI and Environ-,\nmental Impact,\nArea,Policy Proposal and Description\nEnergy and Environmental,Energy consumption from inferences: Include energy consumption from\n,both single and cumulative inferences in reporting.\nReporting,\n,Indirect emissions and water consumption: Mandate reporting of indirect\n,emissions and water use in AI applications.\n,Cumulative", + "type": "table" + }, + { + "rank": 19, + "score": 2.88972628068607, + "content": "0,1\nand downstream providers) through delegated acts from the,\nCommission (Articles 53(5) and (6)) and future recommenda-,\"7.3\nTransparency and Accountability\"\ntions from the AI Office.,Mechanisms\n• Indirect Emissions and Water Consumption Reporting:,To promote public trust and accountability in AI’s environmental\nThe Act currently omits indirect emissions from AI applica-,\"impact, the following m", + "type": "table" + }, + { + "rank": 20, + "score": 2.841915243272355, + "content": ", both\",\n,both increase transparency and facilitate more sustainable\nsingle and overall inferences should be included as a report-,\n,decisions by market participants.\ning category in Annex XI and Annex XII (vis-à-vis authorities,\nand downstream providers) through delegated acts from the,\nCommission (Articles 53(5) and (6)) and future recommenda-,\"7.3\nTransparency and Accountability\"\ntions from the", + "type": "table" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/erben2023_chunks_ranked.json b/wattbot2025/data/ranked/erben2023_chunks_ranked.json new file mode 100644 index 00000000..4c7a7c36 --- /dev/null +++ b/wattbot2025/data/ranked/erben2023_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.753906277995125, + "content": "e-,\n,cloud provider to see the impact on cost and throughput.\nsources cost-effectively and have additional reliability. In our sce-,\n,(1) No inter-cloud throughput penalty. Figure 10 shows the\n\"nario, we are interested in what throughput per $ can be expected\",\n,throughput and granularity of each multi-cloud experiment. CV and\n\"and if any barriers prevent multi-cloud training. However, one can\",\n,", + "type": "table" + }, + { + "rank": 2, + "score": 4.4825239094888225, + "content": "ibuted training setups.\n5 MULTI-CLOUD PERFORMANCE\nUsing multiple cloud providers makes sense if we want to use re-\nsources cost-effectively and have additional reliability. In our sce-\nnario, we are interested in what throughput per $ can be expected\nand if any barriers prevent multi-cloud training. However, one can\nalso consider the data center’s carbon footprint, which can change\ndepending on th", + "type": "text" + }, + { + "rank": 3, + "score": 4.0331160625179665, + "content": "me should increase, and the US-EU communication\",\"cheaper than the DGX-2, and 8xT4, which is 58% cheaper than DGX-\"\nbottleneck should slow us down to the same extent as the E-B-1,\"2, while being 37% slower (Figure 1). The CV model can be scaled\"\n\"experiment. This reduction is a Hivemind-specific anomaly, as it\",\"more easily due to its initially high granularity, which makes the very\"\n\"uses a singl", + "type": "table" + }, + { + "rank": 4, + "score": 4.003707094656505, + "content": "is case, we study the throughput of experiments\nwith resources in the us-west andeu-central regions (B-2,4,6,8).\nThe B-2 experiment has one VM in the US and one in the EU,\nachieving a virtually identical throughput of 68.4 (US-EU) versus 70.1\n(US) at CV (Figure 8a). Our maximum peak egress rate of 250 Mb/s\ndoes not affect the CV experiments, while the US experiments peaked\nat 1.1 Gb/s. The reducti", + "type": "text" + }, + { + "rank": 5, + "score": 4.003707094656505, + "content": "e our band-\"\n\"does not affect the CV experiments, while the US experiments peaked\",width measurements were 210 and 130 Mb/s from the US to the EU\n\"at 1.1 Gb/s. The reduction in bandwidth penalizes NLP harder, where\",\"and ASIA, respectively (Table 3), this suggests that the averaging\"\nwe are 16% slower with 177.3 SPS (US-EU) compared to the intra-zone,was done over the US node and not an N-to-N all", + "type": "table" + }, + { + "rank": 6, + "score": 3.932027487488362, + "content": "more peers. Let us\ncompare the granularity of the experiments for E-B (Figure 13b),\nwhich uses T4 GPUs in the US as an additional cloud resource. Both\nthe computation and communication time decrease with the number\nof GPUs, even increasing the granularity from 1.98 at E-B-2 to 2.15\nat E-B-4. This is surprising since, usually, with more peers, the com-\nmunication time should increase, and the US-EU", + "type": "text" + }, + { + "rank": 7, + "score": 0.0, + "content": "How Can We Train Deep Learning Models Across\nClouds and Continents? An Experimental Study\nAlexander Erben\nTechnical University of Munich\nalex.erben@tum.deRuben Mayer\nUniversity of Bayreuth\nruben.mayer@uni-bayreuth.deHans-Arno Jacobsen\nUniversity of Toronto\njacobsen@eecg.toronto.edu\nABSTRACT\nThis paper aims to answer the question: Can deep learning models\nbe cost-efficiently trained on a global mar", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "e compare the scalability potential for hybrid-cloud scenarios by\nadding cloud resources to on-premise hardware to improve training\nthroughput. Finally, we show how leveraging spot instance pricing\nenables a new cost-efficient way to train models with multiple cheap\nVMs, trumping both more centralized and powerful hardware and\neven on-demand cloud offerings at competitive prices.\nPVLDB Reference F", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "whether to invest in on-premise hardware or move to the\ncloud for deep learning (DL) is not easy. Wanting to scale existing\ninfrastructure means paying upfront, as combining cloud and on-\npremise is not an option with popular DL frameworks due to needing\na dedicated high-bandwidth interconnect. To enabled model- and\ndata-parallelism, current state-of-the-art accelerators have band-\nwidths of 900 G", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "educed rate, typically at\na 40-90% discount (Section 1), but with the drawback that the VM\ncan be terminated at any time if another customer is willing to pay\nthe on-demand price [ 33]. Unfortunately, popular DL frameworks\nhave not been developed with failure semantics in mind and cannot\nadequately deal with peers that fail [ 12]. While services like Amazon\nSagemaker [ 14] and projects like Skypil", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "4.0/ to view a copy of\nthis license. For any use beyond those covered by this license, obtain permission by\nemailing info@vldb.org. Copyright is held by the owner/author(s). Publication rights\nlicensed to the VLDB Endowment.\nProceedings of the VLDB Endowment, Vol. 17, No. 6 ISSN 2150-8097.\ndoi:10.14778/3648160.3648165Table 1: Average us-west cloud pricing in April ’23.\nTypeCloudGC AWS Azure\nT4 Spo", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "fic (inter-region) OCE 0.08 $/GB 0.01 $/GB 0.08 $/GB\nTraffic ANY-OCE 0.15 $/GB 0.02 $/GB 0.08 $/GB\nTraffic (between continents) 0.08 $/GB 0.02 $/GB 0.02 $/GB\n2\n4\n6\n8\n10\nCost in $ per 1M Samples\n0\n500Samples per Second\nDGX-2 DGX-2\n8xT4 8xT4\n1xT4 1xT48xA10\n1xA10DDP 4xT4 DDP 4xT4Instance Type\nSpot\nOn-Demand\nFigure1:CosttothroughputtradeoffforConvNextLargeatdif-\nferent instance types. Our training set", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "aining, Hivemind [ 39], which inherently deals with peers\nthat can stop running at any time. While there is research on how\nHivemind can be used for training on spot VMs [ 17,37,38], it does not\ncompare the cost-throughput tradeoff for different cloud offerings or\nperform ablation studies on geographic distribution or model sizes.\nTo motivate this new possibility, we trained the ConvNextLarge\nmode", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "setup. The single node (1xT4,\n1xA10, DGX-2) experiments show the current state-of-the-art cost-\nthroughput ratio for training on GC and LambdaLabs. The DGX-2\nnode is the fastest, with a throughput of 413 SPS, but it also costs\n$6.30/h ($4.24/1M samples), shown by the horizontal and vertical\nlines. The single-accelerator experiments (1xT4, 1xA10) have a better\ncost-throughput ratio ($0.62/1M sample", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "2306.03163v4 [cs.LG] 2 Jun 2024\n\n(8xT4, 262 SPS, $1.77/1M samples) than using the DGX-2. Every\ncloud provider deals differently with how they price spot instances\nand network traffic (cf. Section 1) and has varying interruption rates\nfor different accelerators [ 23]. Being able to choose the best option\nwas not possible before, and having the option to combine older,\nmore available GPUs is a net", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "istributed spot training\nbecomes viable, what hardware can be used for it, and what the\nminimum bandwidth and latency are. We close this research gap by\nperforming a comprehensive analysis of multiple DL tasks from CV\nand NLP, breaking down how time is spent in each epoch, and com-\nparing them to non-distributed runs to quantify the advantages and\ndisadvantages of distributed spot training. We det", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "through training on up to four continents. For comparison\nof the models’ scalability and to show which of them can be trained\nin a distributed fashion, we introduce the granularity metric , the ratio\nof calculation to communication time, and show how it can be used\nfor predicting performance with different hardware setups. Finally,\nwe summarize our lessons on how to design geo-distributed spot\ntra", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "While we find perfor-\nmance penalties due to remote versus on-premise compute\nresources, the throughput still scales with increased comput-\ning power. By leveraging multiple spot instances with one\nT4 GPU each, we can be more cost-efficient than a DGX-2\nnode or the very competitively priced A10 offerings from\nLambdaLabs.\n(2)We investigate the suitability of geo-distributed train-\ning for various C", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "or effective training. This enables,\nfor the first time, distributed training of smaller million-\nparameter models (12M-560M) over <1 Gb/s bandwidth and\n>150ms latency networks.\n(3)We evaluate two different hybrid-cloud experimental\nsetups with consumer- and server-grade on-premise\nhardware and try to improve the throughput with a band-\nwidth of, at worst, 50 Mb/s to the cloud resources. While we\n", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "c to compare model suitability for distributed\nspot training and estimate training performance with ad-\nditional spot VMs. This provides guidance on the trade-off\nbetween performance and cost when using geo-distributed\nspot instances. To apply our findings, we perform a case-\nstudy on a state-of-the-art model from the ASR domain and\nachieve speedups on low-end hardware.\n2 DEEP LEARNING ON SPOT INS", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/fernandez2025_chunks_ranked.json b/wattbot2025/data/ranked/fernandez2025_chunks_ranked.json new file mode 100644 index 00000000..73901fe9 --- /dev/null +++ b/wattbot2025/data/ranked/fernandez2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 6.7239019764643135, + "content": "model parallelism (Narayanan et al., 2021;\nHuang et al., 2019; Li et al., 2020), speculative\ndecoding (Liu et al., 2024; Leviathan et al., 2023;\nChen et al., 2023, 2025), and disaggregated serving\n(Zhong et al., 2024).\nSolely optimizing system performance for speed\nis insufficient in characterizing and does not pro-\nvide insight into the model energy use and result-\ning carbon emissions of LLM inf", + "type": "text" + }, + { + "rank": 2, + "score": 4.950277393110804, + "content": "y\nand generative Artificial Intelligence (AI) work-,\n,consumption (green) as compared with an unoptimized\n\"loads,\nincluding conversational AI and code\",\n,baseline PyTorch (purple) implementation.\ngeneration. We introduce a modeling approach,\nthat approximates real-world LLM workflows,\n,\"2025). However, the growing prevalence of LLMs\"\nthrough a binning strategy for input-output to-,\n,yields commens", + "type": "table" + }, + { + "rank": 3, + "score": 4.90833003467888, + "content": "icators (Dehghani et al., 2022). Recent work has\",examine a variety of optimization techniques and\nexplored methods for explicitly reducing energy re-,evaluate on representative data corresponding to\nquirements and carbon emissions for LLM serving,classical NLP tasks as well as modern LLM de-\nvia disaggregated serving over heterogeneous hard-,ployment settings. We conclude that the effective-\n\"war", + "type": "table" + }, + { + "rank": 4, + "score": 4.7088233357878275, + "content": "Sophia,\n*Equal contribution\nBurstGPT Azure Code Azure Conv.\nTask101102103Energy (kWh)\nTheoretical\nPyTorch\nvLLMFigure 1: Proper application of efficiency methods with\noptimized vLLM (orange) approaches the ideal energy\nconsumption (green) as compared with an unoptimized\nbaseline PyTorch (purple) implementation.\n2025). However, the growing prevalence of LLMs\nyields commensurate increases in the ener", + "type": "text" + }, + { + "rank": 5, + "score": 4.7088233357878275, + "content": "athan et al., 2023;\",\n,ence energy use often rely on simplified deploy-\n\"Chen et al., 2023, 2025), and disaggregated serving\",\n,ment settings with limited sets of model architec-\n\"(Zhong et al., 2024).\",\n,tures and serving frameworks.\nSolely optimizing system performance for speed,\n\"is insufficient\nin characterizing and does not pro-\",\n,\"6\nConclusion\"\nvide insight into the model energy use and res", + "type": "table" + }, + { + "rank": 6, + "score": 4.652095837218008, + "content": "\"to induce shorter sequence generations (Li et al.,\",ware framework implementations; and that opti-\n\"2024). However, the exact impact or improvements\",mizations cannot be applied uniformly.\nin energy requirements for latency-optimized meth-,\"Additionally, we conduct a case study of classi-\"\nods remains not fully characterized.,cal NLP tasks and real-world LLM inference work-\n,loads and find that p", + "type": "table" + }, + { + "rank": 7, + "score": 4.5785516936302955, + "content": "rogeneous hard-\nware (Shi et al., 2024), system-wide scheduling\nand request routing to energy-optimized instances\n(Stojkovic et al., 2024b), and prompt directives\nto induce shorter sequence generations (Li et al.,\n2024). However, the exact impact or improvements\nin energy requirements for latency-optimized meth-\nods remains not fully characterized.\nEstimations and Measurement of of Energy Use\nin N", + "type": "text" + }, + { + "rank": 8, + "score": 4.489828091041859, + "content": "died\ninference optimizations can reduce total energy use\nby up to 73% on the BurstGPT chat dataset.\nLimitations and Risks\nIn this work, we evaluate the energy efficiency and\ncarbon emissions of LLM inference as approxi-\nmated by total GPU power usage. Although GPUs\nthe majority of arithmetic operations required for\n9\n\ninference and operate at a higher TDP than other\ncomponents, we do not account f", + "type": "text" + }, + { + "rank": 9, + "score": 4.2961680981524335, + "content": "Energy Considerations of Large Language Model Inference and Efficiency\nOptimizations\nJared Fernandez*1, Clara Na*1, Vashisth Tiwari*1,\nYonatan Bisk1,Sasha Luccioni2,Emma Strubell1\n1Carnegie Mellon University,2Hugging Face,\nCorrespondence: {jaredfern, clarana, vashisthtiwari}@cmu.edu\nAbstract\nAs large language models (LLMs) scale in size\nand adoption, their computational and environ-\nmental costs c", + "type": "text" + }, + { + "rank": 10, + "score": 4.240619783802857, + "content": "a RTX A6000 Ada 300W 91.1 –\n128xAMD EPYC 7763 1TB Nvidia RTX A100-80 GB 300W 156 312\nTable 5: Node Hardware Specifications\n2123252729\nBatch size−12−10−8−6−4−202Energy Reduction (%)\n(a) A100 80GB PCIe\n202122232425262728\nBatch size−12.5−10.0−7.5−5.0−2.50.02.55.0Energy Reduction (%) (b) A6000 Ada\n202122232425262728\nBatch size−10.0−7.5−5.0−2.50.02.55.0Energy Reduction (%) (c) A6000\nFigure 9: Energy re", + "type": "text" + }, + { + "rank": 11, + "score": 3.2240345259601644, + "content": "ity and industry as the,\n,Limitations and Risks\nscale of models and prevalence of deployment has,\n\"increased (Schwartz et al., 2020; Wu et al., 2022).\",\"In this work, we evaluate the energy efficiency and\"\nEstimations of the energy requirements and envi-,carbon emissions of LLM inference as approxi-\nronmental impact of LLMs has largely focused on,mated by total GPU power usage. Although GPUs\nestim", + "type": "table" + }, + { + "rank": 12, + "score": 2.167322860935122, + "content": "Maud Texier, and Jeff Dean.\",\n,gpt: A real-world workload dataset to optimize llm\n\"2022.\nThe\ncarbon\nfootprint\nof machine\nlearn-\",\n,\"serving systems. Preprint, arXiv:2401.17644.\"\n\"ing training will plateau,\nthen shrink.\nPreprint,\",\narXiv:2204.05149.,\n,\"Grant Wilkins, Srinivasan Keshav, and Richard Mortier.\"\n\"Jack W Rae, Anna Potapenko, Siddhant M Jayakumar,\",2024. Offline energy-optimal llm serving", + "type": "table" + }, + { + "rank": 13, + "score": 2.15798189096485, + "content": "0,1\n\"Deepak Narayanan, Mohammad Shoeybi, Jared Casper,\",\"Tianyao Shi, Yanran Wu, Sihang Liu, and Yi Ding. 2024.\"\n\"Patrick LeGresley, Mostofa Patwary, Vijay Kor-\",Greenllm: Disaggregating large language model serv-\n\"thikanti, Dmitri Vainbrand,\nPrethvi Kashinkunti,\",ing on heterogeneous gpus for lower carbon emis-\n\"Julie Bernauer, Bryan Catanzaro, et al. 2021.\nEf-\",sions. arXiv preprint arXiv:2412.2", + "type": "table" + }, + { + "rank": 14, + "score": 2.1304359194268994, + "content": "rating the science of lan-,arXiv:2406.14066.\nguage models. arXiv preprint arXiv:2402.00838.,\n,\"Alexandra Sasha Luccioni, Sylvain Viguier, and Anne-\"\n\"Daya Guo, Dejian Yang, Haowei Zhang, Junxiao Song,\",Laure Ligozat. 2023. Estimating the carbon footprint\n\"Ruoyu Zhang, Runxin Xu, Qihao Zhu, Shirong Ma,\",\"of bloom, a 176b parameter language model. Journal\"\n\"Peiyi Wang, Xiao Bi, et al. 2025. Deepseek", + "type": "table" + }, + { + "rank": 15, + "score": 2.0774011527622815, + "content": ", Adam Paszke, Jeff Smith,\nBrian Vaughan, Pritam Damania, et al. 2020. Pytorchdistributed: Experiences on accelerating data parallel\ntraining. arXiv preprint arXiv:2006.15704 .\nXiaoxuan Liu, Cade Daniel, Langxiang Hu, Woosuk\nKwon, Zhuohan Li, Xiangxi Mo, Alvin Cheung,\nZhijie Deng, Ion Stoica, and Hao Zhang. 2024.\nOptimizing speculative decoding for serving large\nlanguage models using goodput. arXi", + "type": "text" + }, + { + "rank": 16, + "score": 2.060304841543708, + "content": "ference with,\n,\"Fan, et al. 2024. The llama 3 herd of models. arXiv\"\nsarathi-serve. Proceedings of 18th USENIX Sympo-,\n,preprint arXiv:2407.21783.\nsium on Operating Systems Design and Implementa-,\n\"tion, 2024, Santa Clara.\",\n,\"Ahmad Faiz, Sotaro Kaneda, Ruhan Wang, Rita Osi,\"\n,\"Prateek Sharma, Fan Chen,\nand Lei\nJiang. 2023.\"\n\"Jordan Aljbour, Tom Wilson, and P Patel. 2024. Power-\",\n,Llmcarbon: Mode", + "type": "table" + }, + { + "rank": 17, + "score": 2.0518617761646016, + "content": "d car-\"\n\"agement opportunities\nfor\nllms\nin the cloud.\nIn\",\n,\"bon considerations of fine-tuning BERT.\nIn Find-\"\n\"Proceedings of\nthe 29th ACM International Con-\",\n,\"ings of\nthe Association for Computational Linguis-\"\nference on Architectural Support for Programming,\n,\"tics: EMNLP 2023, pages 9058–9069, Singapore.\"\n\"Languages and Operating Systems, Volume 3, pages\",\n,Association for Computational Lin", + "type": "table" + }, + { + "rank": 18, + "score": 2.0434876272085716, + "content": "ng library. Advances in\nneural information processing systems , 32.\nPratyush Patel, Esha Choukse, Chaojie Zhang, Íñigo\nGoiri, Brijesh Warrier, Nithish Mahalingam, and Ri-\ncardo Bianchini. 2024. Characterizing power man-\nagement opportunities for llms in the cloud. In\nProceedings of the 29th ACM International Con-\nference on Architectural Support for Programming\nLanguages and Operating Systems, Vol", + "type": "text" + }, + { + "rank": 19, + "score": 2.0351815543110967, + "content": "InInternational Conference on Learning Representa-\ntions .\nAbhimanyu Dubey, Abhinav Jauhri, Abhinav Pandey,\nAbhishek Kadian, Ahmad Al-Dahle, Aiesha Letman,\nAkhil Mathur, Alan Schelten, Amy Yang, Angela\nFan, et al. 2024. The llama 3 herd of models. arXiv\npreprint arXiv:2407.21783 .\nAhmad Faiz, Sotaro Kaneda, Ruhan Wang, Rita Osi,\nPrateek Sharma, Fan Chen, and Lei Jiang. 2023.\nLlmcarbon: Modeling th", + "type": "text" + }, + { + "rank": 20, + "score": 2.0187703429986863, + "content": "erence. In\n2023 IEEE High Performance Extreme Computing\nConference (HPEC) , pages 1–9. IEEE.\nRoy Schwartz, Jesse Dodge, Noah A Smith, and Oren\nEtzioni. 2020. Green ai. Communications of the\nACM , 63(12):54–63.\nArman Shehabi, Alex Hubbard, Alex Newkirk, Nuoa\nLei, Md Abu Bakkar Siddik, Billie Holecek, Jonathan\nKoomey, Eric Masanet, Dale Sartor, et al. 2024. 2024\nunited states data center energy usag", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/griggs2024_chunks_ranked.json b/wattbot2025/data/ranked/griggs2024_chunks_ranked.json new file mode 100644 index 00000000..67e49906 --- /dev/null +++ b/wattbot2025/data/ranked/griggs2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 5.254939359829552, + "content": "nsistently reducing overall cost.\n•Long-context Dataset (PubMed). In Figs. 11b and 11e, Mélange achieves 15-33% cost reduction\n(120ms SLO) and 2-22% reduction (40ms SLO). A100 generally achieves higher T/$for the\nrequest sizes in PubMed, evidenced by the 120ms setting where A100-only is consistently cheaper\nthan H100-only. However, when SLO tightens to 40ms, H100 is the clear winner due to H100’s\n", + "type": "text" + }, + { + "rank": 2, + "score": 4.533083790047353, + "content": "share of\"\n\"A100s at a looser SLO, and more H100s as the SLO is tightened.\"\n\"• Mixed-context Dataset.\nIn Figs. 11c and 11f, Mélange achieves 13-51% cost reduction (120ms\"\n\"SLO) and 4-51% reduction (40ms SLO). Compared to the PubMed workload, A100-only has much\"\ngreater cost efficiency in the Mixed workload than H100 due to a greater portion of short-context\n\"requests, for which A100 achieves greate", + "type": "table" + }, + { + "rank": 3, + "score": 3.8778074726092457, + "content": "ange)\n(a) Arena, SLO = 120ms.\n1 2 4 8 16 32\nRequest Rate (req/s)0.00.51.01.5Cost (w.r.t Mélange) (b) PubMed, SLO = 120ms.\n1 2 4 8 16 32\nRequest Rate (req/s)012Cost (w.r.t Mélange) (c) Mixed, SLO = 120ms.\n1 2 4 8 16 32\nRequest Rate (req/s)0123Cost (w.r.t Mélange)\n(d) Arena, SLO = 40ms.\n1 2 4 8 16 32\nRequest Rate (req/s)0.00.51.0Cost (w.r.t Mélange) (e) PubMed, SLO = 40ms.\n1 2 4 8 16 32\nRequest Rate", + "type": "text" + }, + { + "rank": 4, + "score": 2.845955243890508, + "content": "tion and concentrates on\"\nreducing LLM deployment costs by choosing cost-effective GPU instance types.\n\"2.2\nMachine Learning with Cloud Resources\"\nRecent studies have explored various strategies for reducing the cost of machine learning (ML) infer-\n\"ence or training. Several focus on utilizing spot instances [42, 12, 53, 11], which is complementary to\"\n\"our work. Other work targets deployment on h", + "type": "table" + }, + { + "rank": 5, + "score": 2.834164061571331, + "content": "heterogeneous resources [ 5,6,30,26,27], but focuses\nprimarily on model training rather than serving. Also, lveraging serverless instances for inference\ncost reduction has been examined in [ 2]. Nonetheless, these prior work predominantly concentrate\non machine learning prior to the advent of LLMs, which we show to have unique characteristics that\nsignificantly impact cost efficiency. More recent ", + "type": "text" + }, + { + "rank": 6, + "score": 2.787960453725974, + "content": "0\n\"• Short-context Dataset (Arena).\nIn Figs. 11a and 11d, Mélange achieves 15-77% cost reduction\"\n\"(120ms SLO) and 9-68% reduction (40ms SLO). For both SLOs, L4/A10G are more cost efficient\"\n\"than A100/H100 at\nlow request rates because they achieve greater utilization. For example, at\"\n\"1-2 req/s, H100 is significantly underutilized and incurs exorbitant costs. However, as the rate\"\n\"increases, L4", + "type": "table" + }, + { + "rank": 7, + "score": 2.7322820916313355, + "content": "how much higher relative costs due to their increased latency, requiring\"\nmore instances to meet the tight deadline. Mélange adapts by allocating more L4/A10G at 120ms\n\"SLO and more A100 at 40ms SLO, consistently reducing overall cost.\"\n\"• Long-context Dataset (PubMed). In Figs. 11b and 11e, Mélange achieves 15-33% cost reduction\"\n(120ms SLO) and 2-22% reduction (40ms SLO). A100 generally achieves", + "type": "table" + }, + { + "rank": 8, + "score": 0.0, + "content": "Mélange: Cost Efficient Large Language Model\nServing by Exploiting GPU Heterogeneity\nTyler Griggs∗\nUC BerkeleyXiaoxuan Liu∗\nUC BerkeleyJiaxiang Yu\nNational University of SingaporeDoyoung Kim\nUC Berkeley\nWei-Lin Chiang\nUC BerkeleyAlvin Cheung\nUC BerkeleyIon Stoica\nUC Berkeley\nAbstract\nLarge language models (LLMs) are increasingly integrated into many online ser-\nvices, yet they remain cost-prohibit", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "scape of GPU types and, within these options, higher cost does not always\nlead to increased performance. Instead, through a comprehensive investigation, we\nfind that three key LLM service characteristics (request size, request rate, SLO)\nstrongly influence GPU cost efficiency, and differing GPU types are most cost\nefficient for differing LLM service settings. As a result, the most cost-efficient\na", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "U allocation for a given\nLLM service. We formulate the GPU allocation task as a cost-aware bin packing\nproblem where GPUs are bins and items are slices of the service workload. Our\nformulation’s constraints account for a service’s unique characteristics, allowing\nMélange to be flexible to support diverse service settings and heterogeneity-aware\nto adapt the GPU allocation to a specific service. Co", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "h engines [ 37,24], chatbots [ 34], and virtual assistants [ 28,47,48]. These services are\noften hosted by deploying models on cloud resources. However, deploying LLMs is expensive. The\nsubstantial size and computational demands of LLMs require the use of costly hardware accelerators,\ntypically GPUs2For example, serving Llama2-70b at BF16 precision requires 2 NVIDIA A100-80GB\nGPUs, which costs ove", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "rowing landscape of hardware accelerators — ranging from\nNVIDIA GPUs [ 33] and AMD GPUs [ 45] to Google TPUs [ 17], CPUs [ 23], and others [ 4] — offers\n∗Equal contribution\n2For brevity, we use “accelerator” and “GPU” interchangeably in this work.\nPreprint. Under review.arXiv:2404.14527v4 [cs.DC] 22 Jul 2024\n\na wide array of choices with varying performance specifications and on-demand cloud cos", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "loud GPUs.\nWe find that GPU cost efficiency is determined by three key LLM service characteristics:\n1.Request Size: An LLM request’s size is made up of its input and output token lengths. For small\nrequest sizes, lower-end GPUs generally produce greater T/$than high-end GPUs.\n2.Request Rate: To maximize utilization, provisioned GPU capacity should align with request\nvolume. At low request rates, s", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": ".\nBecause low-end GPUs generally incur higher latency than high-end GPUs, high-end GPUs are\nrequired for stringent SLOs while low-end GPUs can reduce costs in loose-SLO settings.\nConsider a GPU allocation strategy that integrates each of the three observations above: high-cost\nA100 GPUs handle large requests and meet stringent SLOs, but lower-cost A10G GPUs serve\nsmaller requests ( 1) and looser S", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "are highly dependent on LLM service characteristics. The key\nchallenge, then, is creating a GPU allocation framework that can navigate the diversity of LLM\nservices (request sizes, request rates, latency SLOs) and GPU types to find the optimal GPU allocation.\nFigure 1: Mélange framework.\nWe present Mélange3(Fig. 1), a GPU allocation framework that derives the minimal-cost GPU\nallocation for a give", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "rkload that minimizes cost. This task is a natural application of the\ncost-aware bin packing problem, where bins are GPUs and items are slices of the workload. We\nformulate the problem as an integer linear program (ILP) and efficiently solve with an off-the-shelf\nsolver ( 3). Upon solution, Mélange produces the GPU allocation that can serve the LLM service at\nminimal cost while adhering to the ser", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "se dimensions,\nenabling efficient navigation of heterogeneous GPU types given a service specification. Second,\nMélange is flexible . The inputs ( 1a,1b) can be flexibly modified to include new generations of\nGPUs or alternative definitions of SLO, ensuring Mélange is effective for diverse services. Further, to\nthe best of our knowledge, Mélange is the first GPU allocation framework that utilizes m", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "allocation framework that automatically derives the minimal-cost GPU\nallocation for a given LLM service while satisfying an SLO requirement (§ 5).\n•We evaluate Mélange across four GPU types—NVIDIA L4, A10G, A100, and H100. Mélange\nreduces costs by 9-77% for short-context tasks (interactive chats), 2-33% for long-context tasks\n(document-based), and 4-51% in mixed-context workloads (§ 6).\n2 Related ", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "such as scheduling optimiza-\ntion [ 51,1,46], speculative decoding [ 20,18], kernel optimization [ 8,40] and early exiting [ 41,59].\nAdditional optimizations include quantization [ 10,21,49,50] and sparsification [ 9,52]. Instead of\naltering inference logic, our work assumes a fixed inference engine configuration and concentrates on\nreducing LLM deployment costs by choosing cost-effective GPU inst", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "hlights. Another line of work [ 58,36]\nexplores splitting LLM inference into its two phases (prefill and decode) and performing the two\nphases on separate nodes, perhaps with different GPU types. Our work shows that, even within a\nphase, the best GPU type can change based on LLM service specifications.\n3 Background\n3.1 LLM Request Size Variance\n(a) LLaMA-7B\n85X (b) LLaMA-70B\nFigure 2: Request late", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/han2024_chunks_ranked.json b/wattbot2025/data/ranked/han2024_chunks_ranked.json new file mode 100644 index 00000000..fbae14dc --- /dev/null +++ b/wattbot2025/data/ranked/han2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 8.941777248750252, + "content": "ated with electricity\nconsumption: location-based and market-based [10]. Specifically, location-based carbon emissions refer\nto the physical carbon emissions attributed to an electricity consumer connected to the power grid, while\nmarket-based carbon emissions are net emissions after applying reductions due to contractual arrangements\nand other credits (e.g., renewable energy credits). In this pap", + "type": "text" + }, + { + "rank": 2, + "score": 7.12637305417822, + "content": "arbon emissions\"\n\"commonly studied in the literature [8], we focus on criteria air pollutants for AI data centers without con-\"\nsidering market-based pollution reduction mechanisms.\n\"While data centers,\nincluding large technology companies, often use various credits to reduce their\"\n\"market-based carbon emissions [10],\nit\nis likely less effective to apply this practice to mitigate the public\"\n\"hea", + "type": "table" + }, + { + "rank": 3, + "score": 7.033407803917661, + "content": "we use the state-\nlevel data center electricity consumption [5] and run AVERT to calculate the resulting county-level marginal\nair pollutant reduction [70]. AVERT allows a maximum of 15% electricity reduction within an electricity\nregion during each hour. For regions where the data center electricity demand exceeds the 15% reduction\nthreshold for certain hours in 2023, we cap the reduction at 15%,", + "type": "text" + }, + { + "rank": 4, + "score": 5.8657430354407065, + "content": "region. The relationship between the health impact and\nemission reduction in COBRA is approximately linear. Thus, we apply a reduction by x%to the baseline\nemissions of all the power plants within the respective electricity region in COBRA and estimate the corre-\nsponding county-level health impacts, including health outcomes and costs.\nWhen assessing the health impact of generative AI training, w", + "type": "text" + }, + { + "rank": 5, + "score": 5.260661760175559, + "content": "Similarly, to estimate the household electricity bills, we use\"\nthe state-level average price for residential users and county-level average household electricity consumption\nin [92].\nLocation-based emission. There are two types of scope-2 carbon emissions associated with electricity\n\"consumption:\nlocation-based and market-based [10].\nSpecifically,\nlocation-based carbon emissions refer\"\n\"to the ph", + "type": "table" + }, + { + "rank": 6, + "score": 4.565387934161047, + "content": "data collection is 5 minutes. We show in Fig. 5a the region-\nwise normalized interquartile ranges (IQR divided by the yearly average) for both public health costs and\ncarbon emissions. The normalized IQR measures the spread of the time-varying health and carbon signals.\nSpecifically, in 110 out of the 114 U.S. regions (96%), the normalized IQR of health cost is higher than that\nof the carbon inten", + "type": "text" + }, + { + "rank": 7, + "score": 4.565387934161047, + "content": "y average marginal health cost and carbon intensity is 0.292.\n\"2024, provided by [71].6 The time granularity for data collection is 5 minutes. We show in Fig. 5a the region-\"\nwise normalized interquartile ranges (IQR divided by the yearly average) for both public health costs and\ncarbon emissions. The normalized IQR measures the spread of the time-varying health and carbon signals.\n\"Specifically, ", + "type": "table" + }, + { + "rank": 8, + "score": 4.535805641059305, + "content": "inference\nrequests. To date, the existing data centers have mostly exploited such scheduling flexibilities for reducing\nelectricity costs [86], carbon emissions [15], water consumption [87], and/or environmental inequity [88].\nNonetheless, the public health impact of AI significantly differs from these environmental costs or metrics.\nConcretely, despite sharing some common sources (e.g., fossil fu", + "type": "text" + }, + { + "rank": 9, + "score": 4.535805641059305, + "content": "to serve AI inference\n\"requests. To date, the existing data centers have mostly exploited such scheduling flexibilities for reducing\"\n\"electricity costs [86], carbon emissions [15], water consumption [87], and/or environmental inequity [88].\"\n\"Nonetheless, the public health impact of AI significantly differs from these environmental costs or metrics.\"\n\"Concretely, despite sharing some common sourc", + "type": "table" + }, + { + "rank": 10, + "score": 4.217200085760996, + "content": "ectricity apportionment in 2030 may vary from the assumption in AVERT.\nThus, we also consider an alternative apportionment to further evaluate the public health impact of U.S.\ndata centers. Specifically, we consider a state-level electricity apportionment scenario in which each state is\nviewed as an electricity region. The evaluation results are shown in Appendix C and further reinforce our\nkey fi", + "type": "text" + }, + { + "rank": 11, + "score": 4.147411255424593, + "content": "the U.S. national electricity demand, in 2030 [4].\nWhen using McKinsey’s projection, we only use its projected percentage of 11.7%. That is, we consider\nthe EPRI’s projection of non-data center loads and scale up the EPRI’s projection of data center electricity\ndemand to match the percentage of 11.7%. As a result, the 2030 U.S. data center electricity demand is 519\nTWh, instead of 606 TWh, in our ", + "type": "text" + }, + { + "rank": 12, + "score": 3.5615691203001583, + "content": "uce their\nmarket-based carbon emissions [10], it is likely less effective to apply this practice to mitigate the public\nhealth impact. The reason is that, unlike carbon emissions that have a similar effect on climate change re-\ngardless of the emission source locations, the public health impact of criteria air pollutants heavily depends\non the location of the emission source. For example, the publ", + "type": "text" + }, + { + "rank": 13, + "score": 3.368669864538346, + "content": "0\n\"(i.e., the actual health impact of data centers is slightly higher). The county-level emission reduction data\"\nprovided by AVERT is then applied to COBRA to estimate the county-level health outcomes and costs.\n\"Electricity price. When estimating the electricity cost for data centers in 2023 and 2030, we use the state-\"\nlevel average price for industrial users in [92]. The projected U.S. nominal", + "type": "table" + }, + { + "rank": 14, + "score": 3.3147924050653224, + "content": "these losses can be further quantified in economic costs based on epidemiology and economics\nresearch for the corresponding health endpoints [22,42]. In contrast, the environmental impacts of AI, e.g.,\ncarbon emission from fossil fuels and water consumption for data center cooling, often do not cause the\nsame immediate health impacts. For instance, while anthropogenic carbon emissions could also p", + "type": "text" + }, + { + "rank": 15, + "score": 3.2492548711378078, + "content": "(90)\n0200 400 600 8001000\nCarbon (kg/MWh)0306090120Health Cost ($/MWh) (c)Average\nFigure 5: Analysis of marginal scope-2 carbon emission rates and public health costs over 114 U.S. regions between\nOctober 1, 2023 and September 30, 2024 [71]. (a) In 110 out of the 114 U.S. regions (96%), the normalized IQR of\nmarginal health cost is higher than that of marginal carbon intensity. (b) In 90 out of th", + "type": "text" + }, + { + "rank": 16, + "score": 3.238101379977846, + "content": "ater temporal\n\"variation than carbon emissions in 110 out of the 114 U.S. regions. Likewise, in Fig. 5b, the greater temporal\"\nvariation of health costs is also supported by its greater normalized standard deviation (STD divided by the\n\"yearly average) in 90 out of the 114 U.S. regions (79%). Next, we show in Fig. 5c the weak spatial correlation\"\n(Pearson correlation coefficient: 0.292) between th", + "type": "table" + }, + { + "rank": 17, + "score": 3.1854634992937263, + "content": "0\n\"0.0\n0.0\n0\"\n\"0.0\n0.3\n0.6\n0.9\n1.2\n1.5\n0.0\n0.2\n0.4\n0.6\n0.8\n0\n200 400 600 800 1000\"\n\"Carbon IQR\nCarbon STD\nCarbon (kg/MWh)\"\n\"(a) Normalized IQR (110)\n(b) Normalized STD (90)\n(c) Average\"\nFigure 5: Analysis of marginal scope-2 carbon emission rates and public health costs over 114 U.S. regions between\n\"October 1, 2023 and September 30, 2024 [71]. (a) In 110 out of the 114 U.S. regions (96%), the nor", + "type": "table" + }, + { + "rank": 18, + "score": 3.154128326972229, + "content": "0\n\"Yuelin Han\nZhifeng Wu\nPengfei Li\nAdam Wierman\nShaolei Ren1\"\n\"UC Riverside\nUC Riverside\nUC Riverside\nCaltech\nUC Riverside\"\nAbstract\n\"The surging demand for AI has led to a rapid expansion of energy-intensive data centers, impacting the envi-\"\nronment through escalating carbon emissions and water consumption. While significant attention has been\n\"paid to AI’s growing environmental footprint, the ", + "type": "table" + }, + { + "rank": 19, + "score": 2.9791354670811074, + "content": "emporal\nvariation of health costs is also supported by its greater normalized standard deviation (STD divided by the\nyearly average) in 90 out of the 114 U.S. regions (79%). Next, we show in Fig. 5c the weak spatial correlation\n(Pearson correlation coefficient: 0.292) between the yearly average health cost and carbon intensity across\nthe 114 regions. Furthermore, the normalized IQR of the health c", + "type": "text" + }, + { + "rank": 20, + "score": 2.9297063803722887, + "content": "The Unpaid Toll: Quantifying the Public Health Impact of AI\nYuelin Han\nUC RiversideZhifeng Wu\nUC RiversidePengfei Li\nUC RiversideAdam Wierman\nCaltechShaolei Ren1\nUC Riverside\nAbstract\nThe surging demand for AI has led to a rapid expansion of energy-intensive data centers, impacting the envi-\nronment through escalating carbon emissions and water consumption. While significant attention has been\npai", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/jegham2025_chunks_ranked.json b/wattbot2025/data/ranked/jegham2025_chunks_ranked.json new file mode 100644 index 00000000..b44a571c --- /dev/null +++ b/wattbot2025/data/ranked/jegham2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.029396523056855, + "content": "pal sources) or water consumption (the portion of withdrawn water\npermanently lost, primarily through evaporation).\nCIF measures carbon emissions per kilowatt-hour of energy consumed, largely driven by the regional\nelectricity mix. Emissions are categorized as direct on-site combustion (Scope 1), off-site electricity\ngeneration (Scope 2), and embodied emissions from manufacturing and transport (Sc", + "type": "text" + }, + { + "rank": 2, + "score": 3.9085510394954666, + "content": "36]. For\nexample, Scope 1 emissions accounted for only 1.6% of Microsoft’s Scope 2 emissions in 2023 [ 37],\na figure that includes executive air travel, ground transportation, refrigerant leakage, and on-site fuel\nuse, further diminishing the share attributable to data center operations. Accordingly, our analysis\nfocuses exclusively on Scope 2 emissions, which capture the carbon intensity of elect", + "type": "text" + }, + { + "rank": 3, + "score": 3.837926710772746, + "content": "ponsible for evaporating an amount of freshwater equivalent\nto the\"\nannual drinking needs of almost 1.2 million people.\n\"6.4\nEstimated 2025 Annual Carbon Footprint of GPT-4o Inference\"\n\"We further examine GPT-4o’s environmental footprint\nthrough estimated carbon emissions from\"\n\"electricity usage, as seen in Figure 4. Our projections indicate annual emissions of approximately\"\n\"138,125 tons of CO2", + "type": "table" + }, + { + "rank": 4, + "score": 3.7656560675605197, + "content": "d resource consumption during the\ninference phase of the model. Accordingly, embodied emissions and water use from hardware\nmanufacturing and supply chains (Scope 3) are excluded due to their limited relevance to real-time\ndeployment and the risk of inflating per-query estimates when applied without deployment-specific\nattribution or when model lifecycles remain ongoing. For water usage, we focus ", + "type": "text" + }, + { + "rank": 5, + "score": 3.681173067034936, + "content": ", this\nconsumption refers to evaporated freshwater permanently removed from local ecosystems rather than\nrecycled. GPT-4o alone is responsible for evaporating an amount of freshwater equivalent to the\nannual drinking needs of almost 1.2 million people.\n6.4 Estimated 2025 Annual Carbon Footprint of GPT-4o Inference\nWe further examine GPT-4o’s environmental footprint through estimated carbon emissio", + "type": "text" + }, + { + "rank": 6, + "score": 3.487104065846027, + "content": "or long prompts and 0.42 Wh for short ones. Interestingly, GPT-4o mini, although\nsubstantially smaller in parameter count, consumes slightly more energy per query than GPT-4o\ndue to its deployment on less efficient A100 hardware instead of H100s or H200s, illustrating that\ndeployment infrastructure can overshadow model size in determining real-world energy use.\n5.2 Water and Carbon Emissions\nFigur", + "type": "text" + }, + { + "rank": 7, + "score": 3.1326968849728343, + "content": "oc V . Le, Chen Liang, Xinlei Chen, and Andrew Ng.\nCarbon emissions and large neural network training. arXiv preprint arXiv:2104.10350 , 2021.\n[13] Shaolei Li. Making ai less “thirsty”: Uncovering and addressing the secret water footprint of ai\nmodels. arXiv preprint arXiv:2304.03271 , 2023.\n[14] Radosvet Desislavov, Fernando Martínez-Plumed, and José Hernández-Orallo. Trends in ai\ninference energ", + "type": "text" + }, + { + "rank": 8, + "score": 3.124326982568034, + "content": "calability. In parallel, Strubell et al. [ 25] estimated carbon emissions from training\nBERT and GPT-2 by accounting for GPU, CPU, and DRAM power draw alongside PUE adjustments.\nHowever, their analysis excludes inference and infrastructural overhead. Similar limitations appear\nin Meta’s LLaMA reports [ 7,26,27], which provide carbon footprints based on GPUs’ TDPs but\ndisregard water use, system-wi", + "type": "text" + }, + { + "rank": 9, + "score": 3.0664732057104995, + "content": "0,1\n(a) Water consumption per model across three prompt,(b) Carbon emissions per model across three prompt\n\"sizes (ml, log-scale).\",\"sizes (gCO2e, log-scale)\"\nFigure 3: Water consumption and carbon emissions per model.,\n\"Wh (±0.13 Wh), exceeding the footprint of a Google search (0.30 Wh) by approximately 40%.\",\n\"Scaling to a typical daily usage pattern, the cumulative energy reaches 3.73 Wh (±0.35", + "type": "table" + }, + { + "rank": 10, + "score": 2.9723430045701646, + "content": "like OpenAI have a significant advantage in this regard, as their\nhigh traffic volume allows them to rely on higher batch sizes without sacrificing latency to the same\nextent as smaller or less active deployments.\nB Scope 3 Considerations\nWhile this study focuses on operational emissions and resource consumption during inference (Scopes\n1 and 2), it is important to briefly discuss the Scope 3 impa", + "type": "text" + }, + { + "rank": 11, + "score": 2.8934965035183224, + "content": "ds of transatlantic flights and consume water equivalent to the annual\ndrinking needs of millions of people. We revisit this scaling analysis in greater detail in Section 6.\n6 GPT-4o Case Study\n6.1 Energy Cost of a Single GPT-4o User Session\nBased on Reuters [ 68], the average ChatGPT user sends approximately eight queries per day as of\nApril 2025. Based on this, we quantify the per-user energy im", + "type": "text" + }, + { + "rank": 12, + "score": 2.836838206379871, + "content": "and limitations and directions for future work.\n2 Related Work\nThe environmental impact of AI systems has garnered increasing attention in recent years, with a\ngrowing body of work attempting to quantify the energy, carbon, and water costs associated with\ntraining and deploying LLMs.\nLi et al. [ 13] analyzed GPT-3’s freshwater consumption, estimating over 5 million liters used during\ntraining and ", + "type": "text" + }, + { + "rank": 13, + "score": 2.836838206379871, + "content": "0,1\n\"of carbon dioxide and consumes more than 150 milliliters of water per query. For reference, this is\",\n,equivalent to driving 50 meters in a gasoline-powered car and using two-thirds of a standard water\ncup. These figures suggest that environmental impacts are shaped not only by model architecture,\n\"but also by deployment strategies and regional infrastructure conditions. In particular, the el", + "type": "table" + }, + { + "rank": 14, + "score": 2.680301232753991, + "content": "0\nFigure 6: Cross efficiency DEA scores. Bar labels show the AI Index (top) and cross-efficiency score\n(bottom).\n\"manufacturing, emissions from global logistics, and hardware retirement. For instance, Microsoft’s\"\n\"Scope 3 CO2e emissions in 2023 accounted for 66% of the total emissions [17]. Yet, these values\"\n\"are highly variable across vendors, manufacturing locations, and fabrication nodes, and", + "type": "table" + }, + { + "rank": 15, + "score": 2.658504170404558, + "content": "ge in semiconductor\n17\n\nFigure 6: Cross efficiency DEA scores. Bar labels show the AI Index (top) and cross-efficiency score\n(bottom).\nmanufacturing, emissions from global logistics, and hardware retirement. For instance, Microsoft’s\nScope 3 CO 2e emissions in 2023 accounted for 66% of the total emissions [ 17]. Yet, these values\nare highly variable across vendors, manufacturing locations, and fab", + "type": "text" + }, + { + "rank": 16, + "score": 2.557046612781684, + "content": "tigate. As such, sustainable AI deployment must focus on systemic frameworks that\"\n\"assess how well models balance capability with environmental cost. In response, we propose DEA as\"\na principled method for benchmarking model-level eco-efficiency.\n\"7.3\nPolicy Implications\"\n\"As AI\nsystems\nscale globally, ensuring environmental\nsustainability requires both model-level\"\noptimizations and systemic reg", + "type": "table" + }, + { + "rank": 17, + "score": 2.5340501457903923, + "content": "ally\nsought to mitigate. As such, sustainable AI deployment must focus on systemic frameworks that\nassess how well models balance capability with environmental cost. In response, we propose DEA as\na principled method for benchmarking model-level eco-efficiency.\n7.3 Policy Implications\nAs AI systems scale globally, ensuring environmental sustainability requires both model-level\noptimizations and sy", + "type": "text" + }, + { + "rank": 18, + "score": 2.4354856372169884, + "content": "carbon\nemissions per query. We also evaluate eco-efficiency using DEA, mapping sustainability trade-offs\nagainst a composite performance benchmark.\n4.1 Model Selection and Hardware Estimation\nWe analyze 30 large language models across OpenAI, Anthropic, Meta, and DeepSeek. Table 1\nsummarizes each model’s deployment context, including provider, cloud host, hardware type and\nspecifications, and regi", + "type": "text" + }, + { + "rank": 19, + "score": 2.414614760463553, + "content": "ware configurations. We additionally utilize cross-efficiency\nData Envelopment Analysis (DEA) to rank models by performance relative to\nenvironmental cost. Our results show that o3 and DeepSeek-R1 emerge as the\nmost energy-intensive models, consuming over 33 Wh per long prompt, more than\n70 times the consumption of GPT-4.1 nano, and that Claude-3.7 Sonnet ranks\nhighest in eco-efficiency. While a s", + "type": "text" + }, + { + "rank": 20, + "score": 2.3940985500575955, + "content": "er consumption and carbon emissions per model.\nWh (±0.13Wh), exceeding the footprint of a Google search (0.30 Wh) by approximately 40%.\nScaling to a typical daily usage pattern, the cumulative energy reaches 3.73 Wh ( ±0.358Wh). For\nmedium-length queries, this increases to 9.71 Wh ( ±1.106Wh). These results highlight that even\nlimited daily engagement with GPT-4o can impose an energy cost comparab", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/khan2025_chunks_ranked.json b/wattbot2025/data/ranked/khan2025_chunks_ranked.json new file mode 100644 index 00000000..e6b4635c --- /dev/null +++ b/wattbot2025/data/ranked/khan2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 6.748169202121934, + "content": "well with the model’s expectations.\nBelow, we present key examples in Figure. 3 .\nVI. D ISCUSSION\nA. Practical impact\nThe demonstrated reduction in carbon emissions through\noptimization techniques such as quantization and local infer-\nence holds significant value for industries aiming to enhance\nsustainability. With models achieving up to 45% reductions\nin energy consumption, this work directly al", + "type": "text" + }, + { + "rank": 2, + "score": 4.671686987178277, + "content": "practical roadmap for industries and re-\n\"Below, we present key examples in Figure. 3 .\",\n,searchers seeking to balance sustainability with effectiveness.\n,Future research should explore adaptive optimization strategies\nVI. DISCUSSION,\n,\"to minimize\ntrade-offs\nand\ndevelop\nnew metrics\nbalancing\"\n\"A. Practical\nimpact\",\n,\"sustainability with predictive performance. Additionally,\nex-\"\n\"The demonstrate", + "type": "table" + }, + { + "rank": 3, + "score": 3.793484230033921, + "content": "0,1,2,3,4,5\n,,,,\"positive, negative, or neutral). The dataset\",is well-structured\nC. Expected Outcomes,,,,,\n,,,,,\"and contains no missing values, making it highly suitable for\"\n,The proposed framework is expected to significantly reduce,,,,\n,,,,,sentiment analysis tasks in machine learning studies Figure. 2.\nenergy consumption and carbon emissions during LLM infer-,,,,,\n\"ence, while maintaining ac", + "type": "table" + }, + { + "rank": 4, + "score": 3.55111291870638, + "content": "dataset\nis well-structured\"\nC. Expected Outcomes,\n,\"and contains no missing values, making it highly suitable for\"\nThe proposed framework is expected to significantly reduce,\n,sentiment analysis tasks in machine learning studies Figure. 2.\nenergy consumption and carbon emissions during LLM infer-,\n\"ence, while maintaining accuracy and responsiveness compa-\",\n,\"Sentiment Assessment\nInstructions\"\nra", + "type": "table" + }, + { + "rank": 5, + "score": 3.528689015704411, + "content": ", the impact on performance metrics such as accu-\nracy, F1 score, recall, and precision varies. While the reduction\nin carbon footprint is consistent, performance trade-offs are ev-\nident, with some metrics experiencing marginal improvements\n\nand others showing slight declines. For instance, precision\nand recall generally exhibit minor increases in specific cases,\nsuggesting that optimization can ", + "type": "text" + }, + { + "rank": 6, + "score": 3.4469939580184468, + "content": "ing power consumption and utilizing\nemission factor data. Let Edenote the total energy consumed\n(in kWh), and let αbe the emission factor (kg CO 2per kWh).\nWe define the carbon footprint CF as:\nCF=E×α., (4)\nC. Expected Outcomes\nThe proposed framework is expected to significantly reduce\nenergy consumption and carbon emissions during LLM infer-\nence, while maintaining accuracy and responsiveness com", + "type": "text" + }, + { + "rank": 7, + "score": 2.747417758184034, + "content": "gy\nsavings without\ncompromising model\nperformance\",by converting model parameters from high-precision formats\n\"[12]. Likewise, Avatar\nfocuses on creating compact, energy-\",\"(e.g., 32-bit floating-point) to lower-precision formats (e.g., 8-\"\nefficient models optimized for deployment on individual de-,\"bit or even 4-bit),\nthereby reducing memory requirements and\"\n\"vices\n[23]. By\nreducing\ninference\nl", + "type": "table" + }, + { + "rank": 8, + "score": 2.61196874247817, + "content": "e applicability across diverse domains.\n\"sustainability. With models\nachieving up to 45% reductions\",\n,ACKNOWLEDGMENT\n\"in energy consumption,\nthis work directly aligns with corpo-\",\n\"rate\nenvironmental,\nsocial,\nand governance\n(ESG) goals by\",The authors extend their gratitude to the Province of On-\nlowering operational costs and carbon footprints. These tech-,\"tario,\nthe Government\nof Canada\nthrou", + "type": "table" + }, + { + "rank": 9, + "score": 2.412700935271534, + "content": "cy.\n\"illustrating the potential of\ntargeted optimizations [17].\",\"Research efforts have\nshowcased the potential of quanti-\"\n\"In\naddition\nto\nspecific\noptimization\ntechniques,\nbroader\",\"zation\nas\na\nkey\ntechnique\nfor\nenhancing\nenergy\nefficiency\"\n\"frameworks\nfor\nsustainable AI\nhave\nbeen\nproposed. These\",\"in AI\nsystems. For\ninstance, GPTQ (Accurate Post-Training\"\n\"frameworks\nadvocate\nfor\nthe\nintegratio", + "type": "table" + }, + { + "rank": 10, + "score": 2.3254113570669768, + "content": "rating the potential of targeted optimizations [17].\nIn addition to specific optimization techniques, broader\nframeworks for sustainable AI have been proposed. These\nframeworks advocate for the integration of energy-efficient\nalgorithms and the alignment of AI practices with global\nsustainability goals [25]. For instance, intersection of sustain-\nability and software engineering is well establishe", + "type": "text" + }, + { + "rank": 11, + "score": 2.2840930331648703, + "content": "ge (CCS)tCO 2\ncapturedCO2removed and stored\nto prevent releaseIEA, IPCC\nB. Quantization Techniques in LLMs\nQuantization [11] has emerged as a transformative approach\nin optimizing LLMs, addressing the dual challenges of com-\nputational efficiency and environmental sustainability. It works\nby converting model parameters from high-precision formats\n(e.g., 32-bit floating-point) to lower-precision fo", + "type": "text" + }, + { + "rank": 12, + "score": 1.9618389178931184, + "content": "ion techniques. This section reviews,\"kWh\nof electricity consumed\nEnergy\"\n,Agency\n\"key studies that have contributed to this field, highlighting their\",\n,\"Scope 1 Emissions\nDirect\nemissions\nfrom\nGHG Proto-\ntCO2e\"\ncontributions to sustainable AI practices.,\n,\"controlled sources\ncol\"\n\"Efforts to mitigate the environmental\nimpact of LLMs have\",\"Scope 2 Emissions\nIndirect\nemissions\nfrom\nGHG Proto-\ntCO", + "type": "table" + }, + { + "rank": 13, + "score": 1.9428551319219585, + "content": "emissions\nassociated with\nthe CO2\",,,,,,,\n,,,,across value chains,,col,\n\"large-scale models\n[2],\n[4], highlighting significant\nenviron-\",,,,,,,\n,,,tCO2e,,,,\nmental challenges posed by their extensive parameter sizes and,Emissions,,,\"equal\nremovals\",,,\n,Energy,Consump-,MWh,Total energy consumed,,\"IEA, EIA\",\n\"computational demands [15]. Liu and Yin (2024), in particular,\",,,,,,,\n,tion,,,,,,\nemphasiz", + "type": "table" + }, + { + "rank": 14, + "score": 1.9408362322291453, + "content": "0,1\nat high levels even as we adopt energy-saving techniques.,\"actionable\ninsights\nand a\nroadmap for\nadvancing Green AI,\"\n,\"addressing immediate environmental concerns, and fostering\"\nc) Contributions: This study offers several contributions,\n,\"a sustainable future for AI\ntechnologies.\"\n\"to\nthe\ngrowing field\nof Green AI. Primarily,\nit\npresents\na\",\ncomprehensive analysis of carbon emissions generat", + "type": "table" + }, + { + "rank": 15, + "score": 1.7236424239280361, + "content": "ce carbon emissions\n[15]. These\",,,,,,,\n,,,,\"duction or\nremoval\",,Standard,\n\"foundational\ninsights\nunderscore\nthe\nurgency\nof\naddressing\",,,,,,,\n,Carbon,Capture,tCO2,CO2 removed and stored,,\"IEA,\",IPCC\nsustainability in LLM development and deployment.,,and Storage (CCS),captured,\"to prevent\nrelease\",,,\n\"Building on this\nfoundation,\nseveral\ntools and frameworks\",,,,,,,", + "type": "table" + }, + { + "rank": 16, + "score": 1.7227784249535416, + "content": "expressed as\nCO2equivalentIPCC, GHG\nProtocol\nCarbon Intensity gCO 2/\nkWhCO2emissions per unit\nof electricity consumedInternational\nEnergy\nAgency\nScope 1 Emissions tCO 2e Direct emissions from\ncontrolled sourcesGHG Proto-\ncol\nScope 2 Emissions tCO 2e Indirect emissions from\npurchased electricityGHG Proto-\ncol\nScope 3 Emissions tCO 2e Indirect emissions\nacross value chainsGHG Proto-\ncol\nNet Zero\nEmi", + "type": "text" + }, + { + "rank": 17, + "score": 1.7192382991543311, + "content": "0,1,2,3\nMetric,Unit,Definition,Reference\n\"Carbon\nDioxide\nEquivalent\n(CO2e)\",\"Metric\ntons\n(tCO2e)\",\"A\nmeasure\nof\ngreen-\nhouse gases expressed as\nCO2 equivalent\",\"IPCC, GHG\nProtocol\"\nCarbon Intensity,\"gCO2/\nkWh\",\"emissions per unit\nCO2\nof electricity consumed\",\"International\nEnergy\nAgency\"\nScope 1 Emissions,tCO2e,\"Direct\nemissions\nfrom\ncontrolled sources\",\"GHG Proto-\ncol\"\nScope 2 Emissions,tCO2e,\"In", + "type": "table" + }, + { + "rank": 18, + "score": 1.5612465166222433, + "content": "nce preser-\nvation, ensuring that accuracy and responsiveness remain © 2025 IEEE. Accepted to IEEE CAI 2025, to appear in IEEE Xplore.arXiv:2504.06307v1 [cs.LG] 7 Apr 2025\n\nat high levels even as we adopt energy-saving techniques.\nc) Contributions: This study offers several contributions\nto the growing field of Green AI . Primarily, it presents a\ncomprehensive analysis of carbon emissions genera", + "type": "text" + }, + { + "rank": 19, + "score": 1.5612465166222433, + "content": "ron-\",\n,tCO2e\nmental challenges posed by their extensive parameter sizes and,\"Emissions\nequal\nremovals\"\n,\"Energy\nConsump-\nMWh\nTotal energy consumed\nIEA, EIA\"\n\"computational demands [15]. Liu and Yin (2024), in particular,\",\n,tion\nemphasizes the critical role of hardware choices in sustainable,\n,\"Global\nWarming\nRatio\nHeat\ntrapped\nby\na\ngas\nIPCC\"\nAI practices and proposes training methods without com", + "type": "table" + }, + { + "rank": 20, + "score": 1.5347537875893498, + "content": "s, and fostering\na sustainable future for AI technologies.\nA. Carbon Emission Metrics\nMeasuring carbon emissions is essential for understand-\ning and reducing the environmental impact of AI systems.\nVarious metrics have been developed to assess emissions\nacross different scopes, intensities, and stages. This subsection\noutlines widely used metrics, such as Carbon Dioxide Equiva-\nlent (CO 2e), Carb", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/kim2025_chunks_ranked.json b/wattbot2025/data/ranked/kim2025_chunks_ranked.json new file mode 100644 index 00000000..7eef5260 --- /dev/null +++ b/wattbot2025/data/ranked/kim2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 5.277516679047511, + "content": "adford, K. Narasimhan, T. Salimans, and I. Sutskever,\"\napplicable strategy for all workloads and achieves the,“Improving language understanding by generative pre-\n\"greatest\ncost\nreduction when\nselectively\nutilized\nac-\",\n,\"training,” OpenAI Preprint, 2018.\"\n\"cording to workload characteristics.\nIn particular,\nfor\",\"[3] H. Touvron, T. Lavril, G.\nIzacard, X. Martinet, M.-A.\"\n\"offline batch processing", + "type": "table" + }, + { + "rank": 2, + "score": 4.207768615807397, + "content": "s that represent both online chatbot and batch\nprocessing workloads, we were able to derive key insights\nfor the efficient operation of LLM inference systems.\n(i)The impact of a workload’s I/O patterns on optimal\ninfrastructure selection: The requirements of online\nconversational chatbot inference and batch processing\ninference differ greatly in input and output token lengths,\nwhich act as key fac", + "type": "text" + }, + { + "rank": 3, + "score": 0.0, + "content": "Cost-Efficient LLM Serving in the Cloud: VM\nSelection with KV Cache Offloading\nKihyun Kim1, Jinwoo Kim1, Hyunsun Chung1, Myung-Hoon Cha2, Hong-Yeon Kim2, Youngjae Kim1,†\n1Dept. of Computer Science and Engineering, Sogang University, Seoul, Republic of Korea\n2ETRI, Daejeon, Republic of Korea\nAbstract —LLM inference is essential for applications like\ntext summarization, translation, and data analysi", + "type": "text" + }, + { + "rank": 4, + "score": 0.0, + "content": "charac-\nteristics, estimating GPU memory needs, and recommending\ncost-effective VM instances. Additionally, the Compute Time\nCalibration Function (CTCF) improves instance selection\naccuracy by adjusting for discrepancies between theoretical\nand actual GPU performance. Experiments on AWS GPU\ninstances show that selecting lower-cost instances without\nKV cache offloading improves cost efficiency by u", + "type": "text" + }, + { + "rank": 5, + "score": 0.0, + "content": "rn Natural Language Processing (NLP), demon-\nstrating outstanding performance in various applications such\nas text summarization, machine translation, and conversa-\ntional AI [ 1]. LLMs built on Transformer-based architectures,\nsuch as GPT [ 2] and LLaMA [ 3], leverage multi-layer self-\nattention mechanisms and large-scale pretraining to achieve\nnear-human-level language understanding and generati", + "type": "text" + }, + { + "rank": 6, + "score": 0.0, + "content": "s essential to consider task-specific Service Level Objectives\n(SLOs). For instance, in online inference tasks, such as real-\ntime conversational services or question answering, latency\nmust be minimized to ensure a seamless user experience.\nReducing inference latency is a key challenge in these\nscenarios.\nOn the other hand, in batch processing tasks [ 4,5] such\nas text summarization for large dat", + "type": "text" + }, + { + "rank": 7, + "score": 0.0, + "content": "author.can easily lead to GPU memory shortages. Due to the auto-\nregressive nature of LLM inference, the Key-Value (KV) cache,\nwhich stores past token information, continuously grows. As\na result, GPU memory usage increases sharply with sequence\nlength and batch size.\nA common technique to mitigate this issue is KV Cache\nOffloading, which offloads KV cache data exceeding GPU\nmemory limits to CPU m", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "Cloud Environ-\nments: Major cloud service providers such as AWS, GCP, and\nAzure offer a variety of GPU instance options with different\nperformance levels and cost structures, providing flexibility\nin resource utilization [ 10]. However, selecting a cost-efficient\nGPU instance in a cloud environment is a complex task\nthat is difficult for users to perform manually. The challenge\narises because GPU ", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "tion based on task characteristics\n•Efficient KV Cache Offloading strategy\nBalancing throughput targets and cost efficiency by combin-\ning these two factors remains a critical challenge that needs\nto be addressed.\nLimitations of Existing Research: Previous studies on\ncost efficiency in cloud environments [ 11,12,13,14,15] have\nfocused primarily on image processing or general machine\nlearning workl", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "ffloading could be leveraged effectively.\nFurthermore, these studies do not comprehensively analyze\ncost efficiency in relation to Service Level Objectives (SLOs).\nTo address these challenges, this paper proposes InferSave ,\na software framework that automatically selects the optimal\nVM instance by considering both cost and performance based\non SLOs.arXiv:2504.11816v1 [cs.LG] 16 Apr 2025\n\nThe In", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "modeling step to predict the performance\nand cost of each instance. Finally, it evaluates these predictions\nto recommend the most cost-efficient instance that meets the\nuser’s SLO constraints.Through this process, the InferSave\nframework becomes the first solver system that automatically\nrecommends the most economical VM instance for LLM\nserving in cloud environments. By integrating KV cache\nofflo", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "inference in cloud\nenvironments. By leveraging InferSave , users can easily find\nthe most cost-effective VM instance that meets their specified\nSLO while minimizing operational expenses.\nExperimental results show that applying InferSave\nachieves significant cost savings compared to traditional\nmaximum-performance-based policies, with reductions of up\nto 73.7% for online workloads and 20.19% for of", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "as OpenAI’s\nGPT [ 2] and Meta’s LLaMA [ 3], are built on the Trans-\nformer [ 1] architecture. These models consist of a multi-layer\nstructure incorporating Self-Attention mechanisms and Feed-\nForward Networks, enabling their broad applicability across\nvarious natural language processing (NLP) tasks.\nThe LLM inference process is divided into two stages: Prefill\nand Decode. In the Prefill stage, the", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "tored in the\nGPU memory as a Key-Value Cache (KV Cache) to alleviate\ncomputational overhead in subsequent operations.\nThe KV Cache is essential for preventing redundant\ncomputations in Self-Attention, thereby enhancing inference\nspeed and resource efficiency. For instance, if the Prefill stage\ncomputes and stores the Key and Value tensors for the input\n\"I am a,\" the Decode stage can reuse them to ", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "onoperations and improve processing speed. However, the size\nof the KV Cache increases significantly with the input length\nand model size.\nFor example, as shown in Figure 1, in the OPT_2.7B model\nrunning on an AWS g4dn.xlarge instance with 1024 input\ntokens, the KV Cache consumes approximately 0.332GB at a\nbatch size of 2. When the batch size increases to 32, the\nKV Cache expands to 5.312GB, which", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "stion, resulting in an Out-of-\nMemory (OoM) issue. To address this, KV Cache Offloading\ntechniques have been proposed [ 6,7,8,9]. These techniques\noperate by offloading KV Cache data that exceeds GPU\nmemory capacity to CPU memory or disk and retrieving\nit back to the GPU when needed for computation. This\napproach effectively alleviates the GPU memory pressure,\nenabling the processing of long seque", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "tion of KV Cache Offloading. If the transfer\nfrequency of KV Cache data is high, the increased latency\ncan lead to bandwidth bottlenecks, ultimately degrading\ninference performance. Therefore, for effective deployment of\nKV Cache Offloading, it is essential to optimize the process\nby considering LLM inference characteristics (e.g., sequence\nlength, batch size) and user-defined Service Level Object", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "anging from $0.379 (g4ad.xlarge) to $40.96 (p4de.24xlarge),\ndepending on the type of GPU, the memory capacity, and the\nbandwidth of the network [22].\nMoreover, when applying KV Cache Offloading to LLM\ninference, the trade-off between inference performance and\nactual cost introduces a complex dilemma. To maximize cost-\nefficiency, users must carefully optimize their choice of VM\nand offloading stra", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "ly to\n\nTABLE I\nVarious Types of instances provided by AWS.\nThis information was available on Feburary 4, 2025 in\nN.Virginia region.\nNameGPU On- GPU FLOPS vCPU GPU Mem Mem Network\nType Demand ($) (#) (TFLOPS) (GiB) (GiB) (Gbps) (Gbps)\ng4dn.xlarge T4 0.526 1 8.141 4 16 16 - 25\ng4ad.xlarge V520 Pro 0.379 1 7.373 4 8 16 - 10\ng5.xlarge A10G 1.006 1 31.52 4 24 16 - 10\ng5g.xlarge T4G 0.42 1 8.141 4 16 8 ", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "4 32 128 25\ng6.12xlarge L4 4.602 4 30.29 48 96 192 40\ng6.48xlarge L4 13.35 8 30.29 192 196 768 100\np4de.24xlarge A100 40.96 96 19.49 96 7680 640 400\ndetermine an optimal configuration, which adds significant\noverhead [6, 9].\nIn this paper, we outline the key dilemmas of KV Cache\nOffloading for LLM inference in the cloud as follows.\n•Dual Nature of KV Cache Offloading: KV Cache\nOffloading mitigates", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/li2025a_chunks_ranked.json b/wattbot2025/data/ranked/li2025a_chunks_ranked.json new file mode 100644 index 00000000..4a9df32e --- /dev/null +++ b/wattbot2025/data/ranked/li2025a_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 5.5892108423734985, + "content": "important measurement of\na model’s environmental impact (Schwartz et al. 2020) is the\ncarbon footprints originated from the pre-training process.\nWe estimate carbon emission with the methods provided in\n(Patterson et al. 2021). We summarize the carbon footprint\nstatistics of FLM-101B and well-known LLMs in Table 3.\nOur model yields only 1/10 pre-training carbon footprint of a\ntypical LLM.\n4.1 Open", + "type": "text" + }, + { + "rank": 2, + "score": 5.255180043980699, + "content": "model.\n,,\"Going deeper into the nature of these tasks, we further have\"\nCarbon Footprint Analysis. An important measurement of,,\n,,the following observations:\na model’s environmental impact (Schwartz et al. 2020) is the,,\n,,(i) MMLU typically requires domain knowledge to solve.\ncarbon footprints originated from the pre-training process.,,\n,,\"In our training, no English textbook or exam data is int", + "type": "table" + }, + { + "rank": 3, + "score": 4.757312609850994, + "content": "n Table 3.,,\n,,outperforms GLM-130B with only 16B parameters.\nOur model yields only 1/10 pre-training carbon footprint of a,,\n,,\"(ii) As aforementioned, TruthfulQA, ARC, and HellaSwag\"\ntypical LLM.,,", + "type": "table" + }, + { + "rank": 4, + "score": 4.359023474045006, + "content": "0,1,2,3,4,5,6\n\"Table 3: Carbon emissions of our proposed model, FLM-101B, and other well-known LLMs. For details, please see the\",,,,,,\n,\"corresponding references. The definitions of TDP, net tCO2e, and their formulas are the same as (Patterson et al. 2021).\",,,,,\n,GPT-3,Gopher,PaLM,GLM-130B,Llama-2,\nModel,,,,,,FLM-101B\n,(Brown et al. 2020),(Rae et al. 2021),(Anil et al. 2023),(Zeng et al. 2023),(", + "type": "table" + }, + { + "rank": 5, + "score": 3.482558113901408, + "content": "; Klamm, C.; Leong, C.; van Strien, D.; Adelani, D. I.; and\"\n\"with human feedback.\nIn NeurIPS.\",et al. 2022. BLOOM: A 176B-Parameter Open-Access Mul-\n,\"tilingual Language Model. CoRR, abs/2211.05100.\"\n\"Patterson, D.; Gonzalez, J.; Le, Q.; Liang, C.; Munguia, L.-\",\n\"M.; Rothchild, D.; So, D.; Texier, M.; and Dean, J. 2021.\",\"Schwartz, R.; Dodge, J.; Smith, N. A.; and Etzioni, O. 2020.\"\nCarbon emiss", + "type": "table" + }, + { + "rank": 6, + "score": 3.3890286064355553, + "content": "ARC, and HellaSwag\nemphasize more on common sense and Wiki-level knowl-\nedge; their performances improve with the increased amount\nof data and the reduction of training loss. With less than 0.16T\nEnglish data (about 1/10 of Llama-2), FLM-101B already\nachieves the best accuracy of 41.47among all the baselines\non TruthfulQA. On ARC and HellaSwag, FLM-101B is com-\nparable to GLM-130B with a similar a", + "type": "text" + }, + { + "rank": 7, + "score": 3.3890286064355553, + "content": "odels of different sizes (Yao et al. 2024): a smaller model,Figure 2: Training loss for FLM-101B models.\n\"is faster in computation, enabling more rapid consumption\",\nof training data for broader commonsense knowledge; con-,\n\"versely, a larger model is better in the reduction of loss per\",\n,Pipeline Parallel sizes to achieve higher efficiency. The single-\n\"step, indicating a deeper understanding of", + "type": "table" + }, + { + "rank": 8, + "score": 3.3378046474163123, + "content": "Growth (MSG) (Yao et al. 2024), with adaptation.\nSpecifically, to adapt these operators to the multi-node 3D\nparallel framework, we implement them by extending the\nmodel structures offline and reloading the checkpoint when\nthe next stage starts.\nSchedules and Cost-Effectiveness. Model growth schedul-\ning is a trade-off between the pros and cons inherent to\nmodels of different sizes (Yao et al. 202", + "type": "text" + }, + { + "rank": 9, + "score": 3.3378046474163123, + "content": "table. In the 16B stage, 4,608k samples are\nused for learning rate warmup, while in later growth stages,\nwe use fewer samples of 230.4k. Note that we do not apply\nbatch size warmup because we address the stability issue in\na different manner, detailed in Section 3.\n3 Training Stability of FLM-101B\nModels beyond 100B parameters (Scao et al. 2022; Zeng\net al. 2023) usually suffer from a bunch of not", + "type": "text" + }, + { + "rank": 10, + "score": 3.3252396943271054, + "content": "t16 negates the need for loss scale adjustments, making\nour training procedure more promising and reproducible.\nThe full training loss curve is presented in Figure 2. We\nobserve that the loss curve becomes steeper after each growth.\nIt matches the intuition that a larger model is better in loss\nreduction per step. The whole training procedure is robust\nand predictable: even though the 51B stage is", + "type": "text" + }, + { + "rank": 11, + "score": 3.3159697636287877, + "content": "0,1\n\"5School of Computer Science and Engineering, Nanyang Technological University, Singapore\",\nAbstract,Non-Growth vs. three Growth Strategies\n,The shaded area in the graph represents the training cost\nLarge language models (LLMs) are considered important ap-,\n\"proaches towards foundational machine intelligence, achiev-\",\"100\n100\"\ning remarkable success in Natural Language Processing and,80\n,80\n\"", + "type": "table" + }, + { + "rank": 12, + "score": 3.3127689863899024, + "content": ", F.; Miller, L.; Simens, M.;\nAskell, A.; Welinder, P.; Christiano, P. F.; Leike, J.; and Lowe,\nR. 2022. Training language models to follow instructions\nwith human feedback. In NeurIPS .\nPatterson, D.; Gonzalez, J.; Le, Q.; Liang, C.; Munguia, L.-\nM.; Rothchild, D.; So, D.; Texier, M.; and Dean, J. 2021.\nCarbon emissions and large neural network training. arXiv\npreprint arXiv:2104.10350 .\nPenedo, ", + "type": "text" + }, + { + "rank": 13, + "score": 3.143982389029462, + "content": "g remarkable success in Natural Language Processing and\nmultimodal tasks, among others. However, the carbon foot-\nprints and financial costs originating from heavy pre-training\ncomputation is a non-negligible issue. Progressive training\nmethods, inspired by the neurogenesis process that grows\nneural structures, have shown potential to accelerate LLM\npre-training. However, the algorithms, implement", + "type": "text" + }, + { + "rank": 14, + "score": 2.9776617841768895, + "content": "ble, eFLM-16B refers to the professional-knowledge-\nenhanced FLM-16B. Note that C-Eval leaderboard only keeps one decimal place for the evaluation results.\nModel Average Average (Hard) STEM Social Science Humanities Others\nGPT-4 68.7 54.9 67.1 77.6 64.5 67.8\nChatGPT 54.4 41.4 52.9 61.8 50.9 53.6\nGLM-130B 44.0 30.7 36.7 55.8 47.7 43.0\neFLM-16B 46.1 28.9 38.3 53.7 46.8 52.6\nand Chinese is reported t", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "FLM-101B: An Open LLM and How to Train It with $100K Budget\nXiang Li1†, Yiqun Yao1†, Xin Jiang1†, Xuezhi Fang1†, Xuying Meng2,\nSiqi Fan3, Peng Han3, Jing Li4, Li Du1, Bowen Qin1, Zheng Zhang1,\nAixin Sun5, Yequan Wang1∗\n1Beijing Academy of Artificial Intelligence, Beijing, China\n2Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China\n3University of Electronic Science and Tec", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "rations.\nWe believe that further studies on progressive training will ben-\nefit the community by cutting down the costs and promoting\ngreen AI. The checkpoint of FLM-101B is publicly available.\n1 Introduction\nLarge language models (LLMs) (Radford et al. 2018; Tou-\nvron et al. 2023a; Devlin et al. 2019; Raffel et al. 2020) have\nconsistently demonstrated their efficacy across a spectrum of\napplicati", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "ecent trends indicate a shift towards utilizing larger\namounts of data ( e.g., 1.4T tokens for Llama-1 (Touvron et al.\n2023a), 2T tokens for Llama-2 (Touvron et al. 2023b), and\n15T tokens for Llama-3 (Meta 2024)). Meanwhile, the sizes\nof open-sourced models continue to increase (Penedo et al.\n2023; Bi et al. 2024; Mistral 2024). Consequently, a major\nfocus within LLM research is the development of", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "earning (Gong et al.\n2019; Gu et al. 2021; Yao et al. 2024) and neurogenesis\nNon-Growth vs. three Growth Strategies\n(a) Without growthtokens (Trillion)parameters (Billion)\n0204060\n0.75 0.50 0.25 0.00 1.0080100The shaded area in the graph represents the training cost\n(b) Linear growth str ategy cost saving = 50\n(c) Superlinear growth st rateg y cost saving > 50%\n(d) Sublinear growth str ategy ", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "tly\n50%; (c): a superlinear strategy with >50% cost saving; (d):\nsublinear strategy saving the cost by less than 50%.\n(Eriksson et al. 1998). “Growth” means dynamic expansion of\nthe parameter number count, from small to large, through the\ntraining progresses. Figure 1 illustrates three typical growth\nstrategies: linear, sublinear, and superlinear. As the FLOPs\nof LLMs are approximately proportiona", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "tasks un-\nder a fixed FLOPs budget, they mainly consider the scenar-\nios where model sizes are fixed through training. We be-\nlieve that verifying the feasibility of a growth strategy (Gu\net al. 2021; Shen et al. 2022; Chen et al. 2022; Yao et al.\n2024) for extremely large models would be an important\ncompletion to scaling laws. To maximize computational effi-\nciency, we strategically focus on imp", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/li2025b_chunks_ranked.json b/wattbot2025/data/ranked/li2025b_chunks_ranked.json new file mode 100644 index 00000000..3a363823 --- /dev/null +++ b/wattbot2025/data/ranked/li2025b_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 3.421379999305034, + "content": "disproportionately less attention from the AI community as well as the general public. For example, while\nthe scope-2 carbon emissions are routinely included as part of AI model cards, even scope-1 direct water\nusage (either withdrawal or consumption) is missing, let alone scope-2 water usage. This may impede inno-\nvations to enable water sustainability and build truly sustainable AI. Crucially, w", + "type": "text" + }, + { + "rank": 2, + "score": 3.392398826394855, + "content": "ts or crops, or otherwise removed from the im-\"\nmediate water environment” [13]. Water consumption reflects the impact on downstream water availability\nand is crucial for assessing watershed-level scarcity [12].\n\"These two types of water usage correspond to two different water footprints, i.e., water withdrawal foot-\"\n3The scope definition of water usage [8] is in line with that of carbon emission", + "type": "table" + }, + { + "rank": 3, + "score": 3.1876286479778235, + "content": "data center less “thirsty”.\nIn ICAC, 2014.\"\n\"[25] Peter Xiang Gao, Andrew R. Curtis, Bernard Wong, and Srinivasan Keshav.\nIt’s not easy being green.\"\n\"SIGCOMM Comput. Commun. Rev., 2012.\"\n\"[26] Alexandra Sasha Luccioni, Sylvain Viguier, and Anne-Laure Ligozat. Estimating the carbon footprint\"\n\"of BLOOM, a 176B parameter language model.\nJ. Mach. Learn. Res., 24(1), mar 2024.\"\n[27] Microsoft. Micros", + "type": "table" + }, + { + "rank": 4, + "score": 3.0571952908654345, + "content": "ons [1].\n\"4\nOur Recommendations\"\nWe provide our recommendations to address AI’s water footprint from the scheduling and policy perspec-\n\"tives, making future AI more environmentally sustainable.\"\n\"4.1\nMore Transparency and Comprehensive Reporting\"\n\"Despite its growing importance, AI’s water footprint has received relatively less attention.\nFor example,\"\nwhile AI model cards routinely include carbo", + "type": "table" + }, + { + "rank": 5, + "score": 3.032379138499444, + "content": "eans the amount\nof water “evaporated, transpired, incorporated into products or crops, or otherwise removed from the im-\nmediate water environment” [13]. Water consumption reflects the impact on downstream water availability\nand is crucial for assessing watershed-level scarcity [12].\nThese two types of water usage correspond to two different water footprints, i.e., water withdrawal foot-\n3The scop", + "type": "text" + }, + { + "rank": 6, + "score": 3.0079626221695923, + "content": "e AI’s computational demand and improve the overall water effi-\nciency, the water consumption per request may decrease in the future. However, the total water consump-\ntion is likely to continue rising due to the growing demand for AI services and the increasing scale of AI\napplications [1].\n4 Our Recommendations\nWe provide our recommendations to address AI’s water footprint from the scheduling an", + "type": "text" + }, + { + "rank": 7, + "score": 2.925516396092383, + "content": "ious cooling system designs to minimize on-site water consumption [4,17,19], these efforts primarily\nfocus on scope-1 water usage while largely overlooking scope-2 impacts. Just as addressing scope-2 carbon\nemissions is important for mitigating climate change, it is equally crucial to address scope-2 water con-\nsumption to reduce AI’s “true water cost”, as noted by the recent U.S. data center ener", + "type": "text" + }, + { + "rank": 8, + "score": 2.836658237832637, + "content": "0\n\"2027, which is more than the total annual water withdrawal of 4 – 6 Denmark or half of the United King-\"\n\"dom.3 Simultaneously, a total of 0.38 – 0.60 billion cubic meters of water will be evaporated and considered\"\n\"“consumption” due to the global AI demand in 2027. Moreover, these global estimates will be exceeded by\"\nthe total water withdrawal and consumption attributed to AI in the U.S. alo", + "type": "table" + }, + { + "rank": 9, + "score": 2.4762781105397367, + "content": "er, and Jeff Dean. Carbon emissions and large neural network training, 2021.\"\n\"[30]\nJovan Stojkovic, Chaojie Zhang, Inigo Goiri, Josep Torrellas, and Esha Choukse. DynamoLLM: Design-\"\n\"ing LLM inference clusters for performance and energy efficiency.\nIn IEEE International Symposium on\"\n\"High-Performance Computer Architecture (HPCA), 2025.\"\n\"[31] Noah Shumba, Opelo Tshekiso, Pengfei Li, Giulia Fant", + "type": "table" + }, + { + "rank": 10, + "score": 2.393524323976088, + "content": "Patterson, Joseph Gonzalez, Quoc Le, Chen Liang, Lluis-Miquel Munguia, Daniel Rothchild,\nDavid So, Maud Texier, and Jeff Dean. Carbon emissions and large neural network training, 2021.\n[30] Jovan Stojkovic, Chaojie Zhang, Inigo Goiri, Josep Torrellas, and Esha Choukse. DynamoLLM: Design-\ning LLM inference clusters for performance and energy efficiency. In IEEE International Symposium on\nHigh-Perfo", + "type": "text" + }, + { + "rank": 11, + "score": 1.3223735040834723, + "content": "0\n\"0\n1.5\n0.20\"\n\"0.1\n0.3\n0.5\n0.7\n0.9\nI\nI\"\n\"E\nE\nU\nU\nD\nD\nN\nN\nR\nR\nU\nU\"\n\"E\nE\nH\nH\nO\nO\nF\nF\nT\nT\"\n\"T\nT\nCarbon (kg/kWh)\nW\nW\nM\nM\"\n\"(a) Carbon/water efficiency\n(b) Hourly carbon/water efficiency\n(c) Hourly energy fuel mixes\"\n\"Figure 2:\n(a) The U.S. eGRID-level scope-2 water consumption intensity factor vs.\ncarbon emission rate [8, 33].\"\n\"The dashed line represents a linear regression model, showing that the e", + "type": "table" + }, + { + "rank": 12, + "score": 1.2800437502036672, + "content": "l, showing that the eGRID-level scope-2 carbon emission and water\nconsumption efficiencies are not aligned. (b)A 5-day snapshot of scope-2 carbon emission rate and water consumption\nintensity in Virginia, starting from April 4, 2022. The values are calculated based on the fuel mixes, carbon emission\nrate and water consumption intensity for each fuel type [8,20,33]. The scope-2 carbon and water eff", + "type": "text" + }, + { + "rank": 13, + "score": 1.2730263904808168, + "content": "print by enabling demand-side\nflexibility.\n4.3 “Follow the Sun” or “Unfollow the Sun”\nTo cut the carbon footprint, it is preferable to “follow the sun” when solar energy is more abundant. Nonethe-\nless, to cut the water footprint, it may be more appealing to “unfollow the sun” to avoid high-temperature\nhours of a day when WUE is high. This conflict can also be shown in Figure 2(a) and Figure 2(b),", + "type": "text" + }, + { + "rank": 14, + "score": 1.2062020026811728, + "content": "ity usage and thus\"\nmay have lower market-based carbon and water footprints.", + "type": "table" + }, + { + "rank": 15, + "score": 1.1637825555649157, + "content": "rs, which can reduce AI’s water footprint by enabling demand-side\"\nflexibility.\n\"4.3\n“Follow the Sun” or “Unfollow the Sun”\"\n\"To cut the carbon footprint, it is preferable to “follow the sun” when solar energy is more abundant. Nonethe-\"\n\"less, to cut the water footprint, it may be more appealing to “unfollow the sun” to avoid high-temperature\"\n\"hours of a day when WUE is high. This conflict can a", + "type": "table" + }, + { + "rank": 16, + "score": 1.022460746034035, + "content": "provided, the estimate likely considers only the GPU energy\nused during token generation.\nTo account for both the prompt phase and the non-GPU energy consumption of servers, we assume a\n5\n\n0.1 0.3 0.5 0.7 0.9\nCarbon (kg/kWh)0481216Water (L/kWh)AKMS\nHIOAMROENWPP\nNYUP\nAZNM(a) Carbon/water efficiency\nMON TUE WED THU FRI1.51.82.12.42.73.0Water (L/kWh)\n0.200.250.300.350.40\nCarbon (kg/kWh) (b) Hourly ca", + "type": "text" + }, + { + "rank": 17, + "score": 1.0135353588854576, + "content": "areas of critical importance,\"\n\"including tackling global challenges such as climate change. On the other hand, many AI models, especially\"\n\"large generative ones like GPT-4, are trained and deployed on energy-hungry servers in warehouse-scale\"\n\"data centers, accelerating the data center energy consumption at an unprecedented rate [1]. As a result,\"\n\"AI’s carbon footprint has been undergoing scrut", + "type": "table" + }, + { + "rank": 18, + "score": 1.0105947571743112, + "content": "truly sustainable AI.\n1 Introduction\nArtificial intelligence (AI) has enabled remarkable breakthroughs in numerous areas of critical importance,\nincluding tackling global challenges such as climate change. On the other hand, many AI models, especially\nlarge generative ones like GPT-4, are trained and deployed on energy-hungry servers in warehouse-scale\ndata centers, accelerating the data center en", + "type": "text" + }, + { + "rank": 19, + "score": 0.9990010259810851, + "content": "mpacts of carbon and water footprints are not substitutable [1,9]. Therefore,\"\nto judiciously achieve a balance between “follow the sun” for carbon efficiency and “unfollow the sun” for\n\"water efficiency, we need to reconcile the potential water-carbon conflicts by using holistic approaches that\"\nare both carbon-efficient and water-wise.\n\"5\nConclusion\"\n\"In this paper, we uncover AI’s water usage a", + "type": "table" + }, + { + "rank": 20, + "score": 0.9904788080213649, + "content": "are routinely included as part of AI model cards, even scope-1 direct water\"\n\"usage (either withdrawal or consumption) is missing, let alone scope-2 water usage. This may impede inno-\"\n\"vations to enable water sustainability and build truly sustainable AI. Crucially, water and carbon footprints\"\n\"are complementary to, not substitutable of, each other for understanding the environmental\nimpacts.\nIn", + "type": "table" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/luccioni2023_chunks_ranked.json b/wattbot2025/data/ranked/luccioni2023_chunks_ranked.json new file mode 100644 index 00000000..c1bad30d --- /dev/null +++ b/wattbot2025/data/ranked/luccioni2023_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 5.868014511232053, + "content": "0,1\n8,Alexandra Sasha Luccioni and Alex Hernandez-Garcia\n4.3,How do the CO2 emissions produced by training ML models evolve over time?\n,\"Some recent analyses have predicted that the carbon emissions of our field will increase in the future, estimating that\"\n,\"achieving further progress on benchmarks such as ImageNet will require emitting thousands of tons of CO2 [50], whereas\"\n,others have predict", + "type": "table" + }, + { + "rank": 2, + "score": 5.575263624533029, + "content": "rks such as ImageNet will require emitting thousands of tons of CO 2[50], whereas\nothers have predicted a plateau in future emissions due to increased hardware efficiency and carbon offsetting [ 37].\nTherefore, one of the goals of our study was to observe the evolution of carbon emissions over time and study whether\nthere are clear trends. Given that the papers from our study span from 2012 to the", + "type": "text" + }, + { + "rank": 3, + "score": 2.2016285585142596, + "content": "f the data centers used for model training (i.e. the overhead used for heating, cooling,\nInternet etc.), as well as the real-time energy consumption of the hardware used for training. We also do not account\nfor carbon offsets and power purchase agreements, which intend to bring computing centers closer to carbon neutrality\nand which are often taken into account by providers of cloud compute in the", + "type": "text" + }, + { + "rank": 4, + "score": 2.1957373340685504, + "content": "r carbon models for similar amounts of energy\nconsumed. This further supports the analysis carried out in Section 4.1, suggesting that the primary energy source used\nfor training ML models has a strong impact on the overall resulting emissions from model training, and that choosing a\nlow-carbon energy grid can play a significant role towards reducing the carbon emissions of ML model training.\nBesi", + "type": "text" + }, + { + "rank": 5, + "score": 2.193012720372219, + "content": "g time. The choice of hardware has a relatively small influence on the large variation of carbon emissions\n\"that we observe in our sample , given that the TDP ranges from 180 W to 300 W, while the carbon emissions span\"\nfrom 105 kgCO2eq to even less than 10 kgCO2eq (see Section A.2 of the appendix for further details). While using\n\"renewable energy can reduce up to 1,000 the carbon emissions for t", + "type": "table" + }, + { + "rank": 6, + "score": 2.1902945533196103, + "content": "electricity grid used, illustrating two parallel groups of models, both exhibiting a largely linear trend,\"\nwith the more carbon intensive models positioned higher than the lower carbon models for similar amounts of energy\n\"consumed. This further supports the analysis carried out in Section 4.1, suggesting that the primary energy source used\"\n\"for training ML models has a strong impact on the over", + "type": "table" + }, + { + "rank": 7, + "score": 2.165512776327596, + "content": "ng), and there are many aspects of the emissions of model training that remain\nunexplored. In sum, there is a need for a more broad and multi-faceted analysis in order to better understand the scale\nand variation of carbon emissions in our community.\nTools and approaches for measuring carbon emissions. Developing standardized approaches for estimating the carbon\nemissions of model training has als", + "type": "text" + }, + { + "rank": 8, + "score": 2.1518388216160518, + "content": "oelectricity, largely deviate from the main trend,\nwith orders of magnitude less carbon emissions compared to models trained using coal and gas. In other words, models\ntrained with low carbon-intensive energy sources, result in much less carbon emissions, ceteris paribus .\nFig. 2. Estimated energy consumed (kWh) and CO 2(kg) by each model in the data set, plotted in a log-log scale. Colors indicat", + "type": "text" + }, + { + "rank": 9, + "score": 2.151606299991194, + "content": "much of the related work in this field has focused on estimating the carbon\nemissions of model training, there are many pieces of other overall carbon footprint of our field which are still missing:\nfor instance, the carbon emissions of tasks such as data processing, data transfer, and data storage [ 28], as well as the\ncarbon footprint of manufacturing and maintaining the hardware used for traini", + "type": "text" + }, + { + "rank": 10, + "score": 2.1266486099350823, + "content": "unt by providers of cloud compute in their carbon accounting [18]. Despite this, the\"\napples-to-apples carbon analysis that we carried out in the current study provides useful insights about the current\n\"state of carbon emissions in our field, as well as how this has evolved over time in the last 9 years.\"\n\"Furthermore, while this study and much of the related work in this field has focused on est", + "type": "table" + }, + { + "rank": 11, + "score": 2.1165702363683367, + "content": "pirical studies on carbon emissions. A large proportion of research has focused on estimating the carbon emissions\nof specific model architectures and/or comparing the carbon emissions of two or more models and approaches. The\n\"first paper to do so was written by Strubell et al., which estimated that the emissions of training and fine-tuning a large\"\n\"Transformer model with Neural Architecture Sea", + "type": "table" + }, + { + "rank": 12, + "score": 2.0859551844732387, + "content": "W, while the carbon emissions span\nfrom 105kgCO 2eq to even less than 10 kgCO 2eq (see Section A.2 of the appendix for further details). While using\nrenewable energy can reduce up to 1,000 the carbon emissions for the same amount of energy used, the remaining\nfactor responsible for the large variation in both energy and carbon emissions in our sample is therefore the training\ntime.\nManuscript pend", + "type": "text" + }, + { + "rank": 13, + "score": 2.081652762359309, + "content": "e carbon footprint of ML model training to date, and provides us with opportunities to analyze it from a variety of\nangles, which we present in Section 4. In the remaining of this section, we describe our method for estimating carbon\nemissions.\nManuscript pending review\n\n4 Alexandra Sasha Luccioni and Alex Hernandez-Garcia\n3.2 Estimating carbon emissions\nThe unit of measurement typically used for ", + "type": "text" + }, + { + "rank": 14, + "score": 2.058430437078683, + "content": "0,1\nCounting Carbon: A Survey of Factors Influencing the Emissions of Machine Learning,11\n\"5.1\nDiscussion of Results\",\nWhile the total carbon footprint of the field of ML is unclear due its distributed nature and the lack of systematic,\n\"reporting of emissions in different settings,\nin the face of the climate crisis,\",it is important for the ML community\n\"to acquire a better understanding of its e", + "type": "table" + }, + { + "rank": 15, + "score": 2.0475701012227288, + "content": "0\n\"Counting Carbon: A Survey of Factors Influencing the Emissions of Machine Learning\n7\"\n\"also shows that models trained with cleaner energy sources, such hydroelectricity, largely deviate from the main trend,\"\n\"with orders of magnitude less carbon emissions compared to models trained using coal and gas. In other words, models\"\n\"trained with low carbon-intensive energy sources, result in much less", + "type": "table" + }, + { + "rank": 16, + "score": 2.0317123151604215, + "content": "of our methodology\nin Section 3. In Section 4 we present our analysis, and we conclude with our proposals for future work, including a\ncentralized hub for reporting the carbon footprint of machine learning..\n2 RELATED WORK\nMeasuring the environmental impact of ML models is a relatively new undertaking, but one that has been gathering\nmomentum in recent years. In the current section, we present sev", + "type": "text" + }, + { + "rank": 17, + "score": 1.9969377244555577, + "content": "ted to be\nover 3.5 million hours (14.8 days with 10,000 GPUs) [ 38]. Obviously, such long training times result in large amounts of\ncarbon emissions, even with lower carbon intensity energy sources. By way of illustration, the model with the longest\ntraining time in our sample would have reduced by about 30 times the carbon emissions had it used the grid with the\nlowest carbon intensity in our sam", + "type": "text" + }, + { + "rank": 18, + "score": 1.9893751599562943, + "content": "0\n\"4\nAlexandra Sasha Luccioni and Alex Hernandez-Garcia\"\n\"3.2\nEstimating carbon emissions\"\nThe unit of measurement typically used for quantifying and comparing carbon emissions is CO2 equivalents. This unit\n\"allows us to compare different sources of greenhouse (GHG) emissions using a common denominator, that of grams of\"\nCO2 emitted per kilowatt hour of electricity generated (gCO2eq/kWh) 1.\nThe am", + "type": "table" + }, + { + "rank": 19, + "score": 1.98187108063474, + "content": "US and China), are on the high end of the carbon spectrum,\nwith emissions of 350 gCO 2eq/kWh and above. On the other end, the countries with the lowest carbon intensity in our\nsample are Canada (which ranges between 1.30 and 52.89 gCO 2eq/kWh, depending on the province) and Spain (which\nhas a single national energy grid with a median carbon intensity of 220.26 gCO 2eq/kWh), but they only represent", + "type": "text" + }, + { + "rank": 20, + "score": 1.9426491714258305, + "content": "more carbon-intensive energy (e.g. coal).\nFor instance, honing in on the central bottom portion of Figure 2, it can be seen that the models trained using\nhydroelectricity (the blue dots) are about two orders of magnitude lower in terms of carbon emissions than models that\nconsumed similar amounts of energy from more carbon-intensive sources such as coal (in brown) and gas (in orange),\ngiven that t", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/luccioni2024_chunks_ranked.json b/wattbot2025/data/ranked/luccioni2024_chunks_ranked.json new file mode 100644 index 00000000..fe10ccdb --- /dev/null +++ b/wattbot2025/data/ranked/luccioni2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 2.7315433860582483, + "content": "fferent results (see [ 1] for\na detailed comparison). It is therefore difficult to systematically compare the carbon footprints of different models.\nExisting tools and studies have also largely focused on the dynamic power consumption (i.e. the electricity necessary for\npowering hardware) and its resulting emissions. However, there have been several proposals to also take into account\nthe embodied", + "type": "text" + }, + { + "rank": 2, + "score": 2.4939909745835944, + "content": "necessary for,\n,\"powering hardware) and its resulting emissions. However, there have been several proposals to also take into account\",\n,the embodied emissions of ML models (i.e. the emissions that can be attributed to the manufacturing of computing,\n,equipment) into carbon emissions estimates. This has been impeded by a lack of transparency from the designers,\n,\"of common computing hardware such ", + "type": "table" + }, + { + "rank": 3, + "score": 2.2752216682960973, + "content": "ion towards the final quantity of carbon emissions. Given the\nincreasing deployment of ML models in the cloud, several studies have therefore looked at cloud-specific ways to reduce\nthe emissions of ML models such as delayed scheduling, workload elasticity and choosing the least carbon-intensive\nelectricity available Chien et al. [6], Dodge et al. [12], Hanafy et al. [19].\nDespite these empirical ", + "type": "text" + }, + { + "rank": 4, + "score": 2.2579566657210575, + "content": "ssions) of different stages of the ML training and deployment cycle, understanding\",\n,\"trade-offs between training and inference emissions patterns, and characterizing the lifetime emissions of ML models,\",\n,\"and we hope that others will be possible in the future, which would require more transparency from model creators\",\n,regarding both the up front (i.e. training) and downstream (i.e. inference", + "type": "table" + }, + { + "rank": 5, + "score": 2.1932751281545495, + "content": "l of comparing energy intensity and carbon\nemissions of models with differing numbers of parameters when applied to different tasks. To address this question,\nwe selected a subset of 3 tasks – text classification, extractive question answering, and summarization – given their\ndiversity and broad applicability in a variety of settings, and compare the 8 zero-shot models of different sizes, based on", + "type": "text" + }, + { + "rank": 6, + "score": 2.138873746258883, + "content": "0-SXM4-80GB GPUs hosted on Amazon Web Services, and\",\n,used the Code Carbon package [47] to measure both the energy consumed and the carbon emitted during inference 3.,\n,\"Given that all of our experiments were run in the same compute region (AWS’s us-west-2), which is based in Oregon\",\n,\"and has an average carbon intensity of 297.6 grams of 𝐶𝑂2𝑒𝑞 per kWh4, this means that both the energy consumed\"", + "type": "table" + }, + { + "rank": 7, + "score": 2.115979415260474, + "content": "times to measure the significance of results.\nWe ran all of our experiments on a node of 8 NVIDIA A100-SXM4-80GB GPUs hosted on Amazon Web Services, and\nused the Code Carbon package [ 47] to measure both the energy consumed and the carbon emitted during inference3.\nGiven that all of our experiments were run in the same compute region (AWS’s us-west-2 ), which is based in Oregon\nand has an average ", + "type": "text" + }, + { + "rank": 8, + "score": 2.104086030820521, + "content": "and Li Fei-Fei. 2017. Visual Genome: Connecting Language and Vision Using Crowdsourced Dense Image Annotations.\"\n\"https://doi.org/10.1007/s11263-016-0981-7\nInternational Journal of Computer Vision 123 (2017), 32–73.\"\n[25] Alex Krizhevsky. 2009. Learning multiple layers of features from tiny images. Technical Report.\n\"[26] Alexandre Lacoste, Alexandra Luccioni, Victor Schmidt, and Thomas Dandres. 2", + "type": "table" + }, + { + "rank": 9, + "score": 2.065006678774233, + "content": "s remains a relatively under-explored topic, albeit one that\nhas been gathering traction since Strubell et al’s seminal article quantifying the energy and carbon emissions of a\nvariety of then-large NLP models [2019]. Since then, most studies have focused on estimating the energy consumed and\ncarbon emitted during the training phase of neural networks – this includes studies by Patterson et al. [2", + "type": "text" + }, + { + "rank": 10, + "score": 2.0224748838677735, + "content": "0,1\n,\"Beyond the differences between task-specific and multi-purpose models generally, we also observed variation\"\nwithin the multi-purpose models that we examined. We present our results in Table 3;,\"in it, we can observe that\"\non a per-architecture basis (i.e. within the family of decoder-only models and the family of sequence-to-sequence,\n,\"models), size and emissions are correlated, with small", + "type": "table" + }, + { + "rank": 11, + "score": 2.0155317623568694, + "content": "e\nand carbon emissions of their products, we can make a comparison based on the experiments carried out in the\npresent study. For instance, the average emissions of a BERT-based model fine-tuned for extractive question answering\n(bert-large-uncased-whole-word-masking-finetuned-squad ), a task akin to extractive web search, is 0.70g 𝐶𝑂2𝑒𝑞\nper 1,000 queries, which is less than 3 times that of the mu", + "type": "text" + }, + { + "rank": 12, + "score": 2.008637314149687, + "content": "0,1,2\nPower Hungry Processing,,\"ACM FAccT ’24, June 3–6, 2024, Rio de Janeiro, Brazil\"\n,\"inference, we hope this study can be useful for practitioners to better understand accuracy-efficiency trade-offs across\",\n,\"tasks and models, as well as enabling better estimates, and projections and policy decisions at the sector level.\",\n\"2\nPREVIOUS WORK\",,\n,\"Estimating the energy and emissions of ML models", + "type": "table" + }, + { + "rank": 13, + "score": 2.0017910165043395, + "content": "as GPT-4 and PaLM being deployed in user-facing\"\n\"products such as web search [4, 18], email, and navigation [17], where smaller, task-specific versions of models such\"\n\"as BERT were previously used [3, 16]. While it\nis hard to quantify the environmental\nimpacts of\nthis transition\"\n\"given the lack of\ntransparency of\ntechnology companies regarding both the number of parameters, architecture\"\n\"and c", + "type": "table" + }, + { + "rank": 14, + "score": 1.9882408199822548, + "content": "erage output length, carbon emissions, and model structures for the different summarization datasets. It\nshows a clear correlation between output length and measured emissions, with a higher slope for the decoder-only\narchitectures (the BLOOMz family of models) than for the sequence-to-sequence architectures (the Flan-T5 family).\nAs we have observed in the current section, there is no ‘one-size-fi", + "type": "text" + }, + { + "rank": 15, + "score": 1.9358666017992041, + "content": "0,1,2\nPower Hungry Processing,,\"ACM FAccT ’24, June 3–6, 2024, Rio de Janeiro, Brazil\"\n\"Fig. 2. The 5 modalities examined in our study, with the number of parameters of each model on the x axis and the average amount\",,\nof carbon emitted for 1000 inferences on the y axis. NB: Both axes are in logarithmic scale.,,\n,\"Next, we examine the respective influences of model size and task structure on mode", + "type": "table" + }, + { + "rank": 16, + "score": 1.9328735854204562, + "content": "least carbon-intensive\",\n,\"electricity available Chien et al. [6], Dodge et al. [12], Hanafy et al. [19].\",\n,\"Despite these empirical studies, there is currently a lack of standardized methodology for quantifying and comparing\",\n,\"the energy consumption and carbon emissions of ML models. There are several tools that exist, such as Code Carbon [47],\",\n,\"MLCO2 [26] and LLMCarbon [13], all of which a", + "type": "table" + }, + { + "rank": 17, + "score": 1.9259800373344897, + "content": "during inference, with differing progressions for each modality – however, the task structure ac-\",,\ncounts for more of the variation than the model size does. We can observe once again that text-to-image is by far the most,,\n\"carbon- and energy-intensive task, with smaller image generation models such as segmind/tiny-sd that have around\",,\n\"500M parameters producing magnitudes more carbon than te", + "type": "table" + }, + { + "rank": 18, + "score": 1.9191367048723835, + "content": "-text tasks, we see two separate sets of models: the masked language modeling task follow-\ning a lower trend, producing emissions akin to text-to-category models, compared to text generation and summarization\ntasks, which produce similar amounts of carbon to the image captioning models with a similar number of parameters.\nFor context, the most carbon-intensive image generation model ( stable-diffu", + "type": "text" + }, + { + "rank": 19, + "score": 1.85971681603551, + "content": "carbon as well as\",\n\"water and mining of rare earth minerals, have yet to be estimated. According to AWS, the largest global cloud provider,\",\n\"inference is estimated to make up 80 to 90% of total ML cloud computing demand [2, 28], whereas a 2021 publication by\",\n\"Meta attributed approximately one-third of their internal end-to-end ML carbon footprint to model inference, with the\",\n\"remainder prod", + "type": "table" + }, + { + "rank": 20, + "score": 1.8533462108760541, + "content": "rding to AWS, the largest global cloud provider,\ninference is estimated to make up 80 to 90% of total ML cloud computing demand [ 2,28], whereas a 2021 publication by\nMeta attributed approximately one-third of their internal end-to-end ML carbon footprint to model inference, with the\nremainder produced by data management, storage, and training [ 57]; similarly, a 2022 study from Google attributed\n", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/luccioni2025a_chunks_ranked.json b/wattbot2025/data/ranked/luccioni2025a_chunks_ranked.json new file mode 100644 index 00000000..d9cc2117 --- /dev/null +++ b/wattbot2025/data/ranked/luccioni2025a_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 9.747596090624022, + "content": "ilities were planned. Meanwhile\"\nenvironmental impacts have been emphasized in research papers,the ongoing resource pressures continue in China and Taiwan.\n\"and scientific reports over the last years [47, 91]. The contribution\",\"Moreover, the race to decouple US semiconductor supply chains\"\nof AI to this expansion is worth considering when measuring the,from geopolitical adversaries has led to int", + "type": "table" + }, + { + "rank": 2, + "score": 7.774602015270526, + "content": "-commerce,\nthe details of which are largely considered sensitive trade secrets,\nthe corresponding machine learning approaches of recommenda-\ntion and ranking are active topics of research in the AI community,\npowered by an enormous quantity of user data being continuously\ngathered through interactions with digital platforms [ 19,20,31].\nThe rapid expansion of online advertising its direct and indi", + "type": "text" + }, + { + "rank": 3, + "score": 5.638022483558511, + "content": "issions budget, we might\"\nand (2) that offsetting rarely mitigates localized impacts to the com-,selectively incentivize uses of AI that contribute towards the UN’s\nmunities where emissions or other environmental degradation is,sustainable development goals [130] or mitigate national security\n\"occurring.\nIncreasingly, companies such as Intel and TSMC are\",\"concerns, over e.g. generating personaliz", + "type": "table" + }, + { + "rank": 4, + "score": 5.353155919964532, + "content": "s” [ 18]. This line of thoughtposits that AI will be subject to the same market pressures, such\nas energy prices, as any other use case, and as a result will ben-\nefit from the same innovations, such as a market transitions to\nrenewable energy. This oversimplification ignores the reality that\n(1) AI is already responsible for non-trivial negative environmental\nexternalities and (2) we can, and lik", + "type": "text" + }, + { + "rank": 5, + "score": 4.811732313453475, + "content": "ction goals of many AI-driven\ncompanies whose revenue models depend on it [135].\nTime rebound effects occur when an innovation changes con-\nsumers’ use of their time, which then frees up (or removes) time\nfor other activities that they carry out. For example, if a vacuum\ncleaner reduces the time needed to clean the floors in a house, it\ncan free up time for leisure activities such as reading or sp", + "type": "text" + }, + { + "rank": 6, + "score": 4.6937532694553585, + "content": "p predict the impacts of geoengineering\non existing ecosystems and avoid possible negative side effects [ 3].\nAI’s negative climate impact is negated by carbon-free energy and\ncarbon offsets. AI-driven products and services are the primary\ncontributor to AI’s direct environmental impacts, and leading AI\ncompanies including Google, Microsoft, Amazon and Meta have\ncommitted to achieving net-zero car", + "type": "text" + }, + { + "rank": 7, + "score": 4.6937532694553585, + "content": "would also require significant investment in infrastructure\nand navigating corresponding sociopolitical systems), PPAs and\ncarbon offsets will remain necessary to fill the gap. However, off-\nsetting was only ever meant to serve as a temporary stop-gap to\nhelp reduce emissions in the short term, and does not represent\na viable replacement to reducing actual emissions. Fundamental\nlimitations to car", + "type": "text" + }, + { + "rank": 8, + "score": 4.403085634110832, + "content": "Mobil that uses AI to expand oil and,\n,help streamline the transition towards renewable energy sources\n\"gas production in Texas and New Mexico by 50,000 barrels of oil per\",\n,\"on a global scale [66, 107] In recent years, the development of AI\"\nday could add up to 640 percent more carbon emissions compared to,\n,for accelerating scientific discovery in materials science has be-\n\"the company’s carbon", + "type": "table" + }, + { + "rank": 9, + "score": 4.151757363190526, + "content": "sible for non-trivial negative environmental\n\"help reduce emissions in the short term, and does not represent\",\"externalities and (2) we can, and likely should, regulate certain\"\na viable replacement to reducing actual emissions. Fundamental,\"uses of AI as appropriate/inappropriate or necessary/unnecessary,\"\n\"limitations to carbon offsetting are:\n(1)\nthe difficulty of proving\",and that this does r", + "type": "table" + }, + { + "rank": 10, + "score": 4.091911738531265, + "content": "harm that they perpetuate. For instance, a\ncoalition of Microsoft employees estimated that a single deal the\ncompany struck with Exxon Mobil that uses AI to expand oil and\ngas production in Texas and New Mexico by 50,000 barrels of oil per\nday could add up to 640 percent more carbon emissions compared to\nthe company’s carbon removal targets for the year [ 119], yet these\nnumbers were not included ", + "type": "text" + }, + { + "rank": 11, + "score": 4.044248137704248, + "content": "xandra Luccioni, Victor Schmidt, and Thomas Dandres.\n2019. Quantifying the carbon emissions of machine learning. arXiv preprint\narXiv:1910.09700 (2019).[68] Pengfei Li, Jianyi Yang, Mohammad A Islam, and Shaolei Ren. 2023. Making\nAI Less\" Thirsty\": Uncovering and Addressing the Secret Water Footprint of AI\nModels. arXiv preprint arXiv:2304.03271 (2023).\n[69] Alexandra Sasha Luccioni, Yacine Jernit", + "type": "text" + }, + { + "rank": 12, + "score": 4.03008737648424, + "content": "e Climate Change 12, 6 (2022), 518–527.\n[64] Jared Kaplan, Sam McCandlish, Tom Henighan, Tom B Brown, Benjamin Chess,\nRewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei. 2020.\nScaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020).\n[65] Jonathan Koomey and Eric Masanet. 2021. Does not compute: Avoiding pitfalls\nassessing the Internet’s energy and carbon im", + "type": "text" + }, + { + "rank": 13, + "score": 4.028383970467848, + "content": "veness, and planetary\"\nlearning approaches to predict electricity production from renewable energy,\"environmental implications. Journal of environmental management 209 (2018),\"\n\"sources. Energies 15, 23 (2022), 9146.\",81–92.\n\"[67] Alexandre Lacoste, Alexandra Luccioni, Victor Schmidt, and Thomas Dandres.\",\"[93] Guillem Ramírez, Matthias Lindemann, Alexandra Birch, and Ivan Titov. 2023.\"\n2019. Quan", + "type": "table" + }, + { + "rank": 14, + "score": 4.004053162607215, + "content": "otprint of machine learning training will plateau,\",\n,While debates over AI’s role in climate change and sustainability\nthen shrink” thanks to continued innovations in machine learn-,\n,\"have become increasingly polarized, both sides have tended to\"\n\"ing models, specialized hardware platforms, data center efficiency,\",\n,focus only on the direct impacts of this technology— positive and\n\"scheduling a", + "type": "table" + }, + { + "rank": 15, + "score": 3.8386544650001033, + "content": "ency,\nscheduling and use patterns, which will reduce overall energy use\nand emissions. The claim that increasing AI efficiency will lead to\nan overall reduction in AI’s resource use is a clear example of why\na deeper engagement with indirect effects is needed in the work\non AI and climate change.\nSimilarly to Jevons’ Paradox, just because an AI model becomes\nmore efficient, that does not imply tha", + "type": "text" + }, + { + "rank": 16, + "score": 3.6332566970274804, + "content": "ironmental impacts of e-commerce.\n[127]\"\nebooks-a-life-cycle-comparison/,\"In International conference on environment Science and engineering, Vol. 8. 202–\"\n[104] Neil Selwyn. 2024. Digital degrowth: Toward radically sustainable education,207.\n\"technology. Learning, Media and Technology 49, 2 (2024), 186–199.\",\"[128] Bill Tomlinson, Rebecca W Black, Donald J Patterson, and Andrew W Torrance.\"\n\"Nvid", + "type": "table" + }, + { + "rank": 17, + "score": 3.6203720396632364, + "content": "lectively incentivize uses of AI that contribute towards the UN’s\nsustainable development goals [ 130] or mitigate national security\nconcerns, over e.g. generating personalized ads for social media\n(as we discussion in Section 3). Finally, AI does differ substantially\nfrom other energy sinks in its potential to engender vast transfor-\nmations in the economy and society, similar to how the advent o", + "type": "text" + }, + { + "rank": 18, + "score": 3.4781837674200187, + "content": "airos\nPower. https://blog.google/outreach-initiatives/sustainability/google-kairos-\npower-nuclear-energy-agreement/\n[125] Neil C. Thompson, Kristjan Greenewald, Keeheon Lee, and Gabriel F. Manso.\n2021. Deep Learning’s Diminishing Returns: The Cost of Improvement is\nBecoming Unsustainable. IEEE Spectrum 58, 10 (2021), 50–55. https://doi.org/\n10.1109/MSPEC.2021.9563954\n[126] JW Thomson. 1972. Method", + "type": "text" + }, + { + "rank": 19, + "score": 3.4342105661471454, + "content": ", and Andrew W Torrance.\n2024. The carbon emissions of writing and illustrating are lower for AI than for\nhumans. Scientific Reports 14, 1 (2024), 3732.\n[129] Alexandra Tremayne-Pengelly. 2024. Amid the A.I. Boom, These States Have\nBecome Data Center Hubs. https://observer.com/2024/12/ai-demand-where-\ndata-centers-using-most-energy/\n[130] UN General Assembly. 2015. Transforming our world : the 203", + "type": "text" + }, + { + "rank": 20, + "score": 3.4342105661471454, + "content": "supply chain\nconstraint significantly narrows the scope of AI’s potential as a,,,\n,,,\"studies, carbon emissions of training large-scale models, energy\"\n\"climate intervention, often leaving only those applications that\",,,\n,,,\"and water consumption, and e-waste from hardware—as well as\"\npromise quick returns or minimal disruptions to existing market,,,\n,,,mapping the ways AI innovations reshape eco", + "type": "table" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/luccioni2025b_chunks_ranked.json b/wattbot2025/data/ranked/luccioni2025b_chunks_ranked.json new file mode 100644 index 00000000..e2381fab --- /dev/null +++ b/wattbot2025/data/ranked/luccioni2025b_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 7.058416545770799, + "content": "eclaration [2023],\"\nillustrating the disconnect between sustainability and ethics in recent approaches to AI regulation.\n\"2Recent media coverage of Microsoft’s sustainability promises has estimated that a single contract to use AI\nto expand oil production “could enable\"\n\"carbon emissions adding up to 640 percent of the company’s carbon removal targets\"\" [191].\"", + "type": "table" + }, + { + "rank": 2, + "score": 6.715940336259481, + "content": "ion, since it takes into account different st ages of the model life cycle including the manufacturing of\ncomputing hardware, idle energy usage, and model deploymen t, finding that training accounted for only half of the\nmodel’s overall emissions [121], meaning that similar stud ies that only took training into account were potentially\nunderestimating their emissions by half. Also, while comme ndabl", + "type": "text" + }, + { + "rank": 3, + "score": 5.8998270468300404, + "content": "hno logical Reality. IEEE Intelligent Systems 30, 03 (may 2015), 2–5.\nhttps://doi.org/10.1109/MIS.2015.53\n[217] Hongyang Zhang, Yaodong Yu, JiantaoJiao, EricXing, L aurentEl Ghaoui, and MichaelJordan.2019. Theoreticallyp rincipled trade-off between\nrobustnessand accuracy.In International conference onmachine learning . PMLR,7472–7482.\n[218] AramZiai.2016. Development discourseand global history:Fro", + "type": "text" + }, + { + "rank": 4, + "score": 5.245938917427876, + "content": "for governing AI. If we take a look at recent\ncommunity endeavors for AI governance, the 2022 Big Science workshop proposeda bottom-upapproach that estab-\nlished mechanisms for various ethical aspects of the projec t such as data governance, quality metrics, and fostering\nstakeholdercollaborationandtransparency[92],aswella sdraftingaconsensus-driven ethicalframeworkforgovern -\ning the resulting ar", + "type": "text" + }, + { + "rank": 5, + "score": 5.110193257891709, + "content": "n endeavors are useful to establish functional mechanisms for governing AI. If we take a look at recent\n\"community endeavors for AI governance, the 2022 Big Science workshop proposed a bottom-up approach that estab-\"\n\"lished mechanisms for various ethical aspects of the project such as data governance, quality metrics, and fostering\"\n\"stakeholder collaboration and transparency [92], as well as dra", + "type": "table" + }, + { + "rank": 6, + "score": 5.110193257891709, + "content": "example,\"\n\"to include sustainability approaches, developers would also provide information on the environmental\nfootprint of\"\n\"running the AI system, such as energy consumption during data processing and potential environmental benefits of\"\n\"the proposed urban layouts, like reduced carbon emissions from optimized traffic flows or green spaces. In this way, by\"\n\"deepening the concept of transparency to", + "type": "table" + }, + { + "rank": 7, + "score": 5.044921228400277, + "content": "ection in the research community with sustainability writ large.\"\n\"Progressing from this observation, while sustainability-oriented research has not been prominently featured in\"\n\"venues that directly address AI ethics,\nit remains a topic of research that has been gathering momentum in recent\"\n\"years. The first research to formally address the environmental\nimpacts of training AI models was the sem", + "type": "table" + }, + { + "rank": 8, + "score": 4.878745834290865, + "content": "hey require, which are unattainable to many members of the AI community, as well as the propagation\"\n\"of biases via their usage. Furthermore, the authors themselves note that there is currently no information available\"\n\"about\nthe embodied emissions linked to manufacturing GPUs, so it\nis impossible to estimate what portion of\nthe\"\noverall carbon footprint this represents. This highlights that the ", + "type": "table" + }, + { + "rank": 9, + "score": 4.514347993782725, + "content": "transparent bu t also understandable in a broader societal context.\nThis augmented view of transparency, which would integrate both social and environmental dimensions, resonates\nwith the increasing awareness that AI is not simply a technol ogical tool but a socio-technical system with extensive\nrepercussions, spanning bothpeopleand theenvironment [2 02].\nEquity.RecognizingthatGreenAI(orsustainabl", + "type": "text" + }, + { + "rank": 10, + "score": 4.348380826430739, + "content": "0\nBridging the Gap:\n\"Integrating Ethics and Environmental Sustainability in AI Research and Practice\n9\"\n\"models, their carbon footprint, is rarely, if ever, disclosed. While model cards of recent models such as BLOOM [214]\"\n\"and Stable Diffusion [169] have included carbon footprint information, it remains far from common information com-\"\nmunicated by model creators – recent work has found that the", + "type": "table" + }, + { + "rank": 11, + "score": 4.306463421602335, + "content": "gmented view of\ntransparency, which would integrate both social and environmental dimensions, resonates\"\nwith the increasing awareness that AI is not simply a technological tool but a socio-technical system with extensive\n\"repercussions, spanning both people and the environment [202].\"\n\"Equity. Recognizing that Green AI (or sustainable AI), i.e. the development and deployment of AI systems that pu", + "type": "table" + }, + { + "rank": 12, + "score": 3.9950907433378577, + "content": "nce it takes into account different stages of the model\nlife cycle including the manufacturing of\"\n\"computing hardware,\nidle energy usage, and model deployment, finding that training accounted for only half of the\"\n\"model’s overall emissions [121], meaning that similar studies that only took training into account were potentially\"\n\"underestimating their emissions by half. Also, while commendable in ", + "type": "table" + }, + { + "rank": 13, + "score": 3.4170688462837893, + "content": "ough its exactorigins areunclear.\n\n8 Alexandra Sasha Luccioni, Giada Pistilli, Raesetje Sefal a, and Nyalleng Moorosi\nSimilarly,evaluating theenvironmental impacts ofAI syst ems is far fromstraightforward,and we arestillmissing\nmany pieces of the puzzle needed in order to meaningfully est imate these impacts. For instance, most of the carbon\nfootprintassessments onlyfocus onthetraining stage ofAI ", + "type": "text" + }, + { + "rank": 14, + "score": 3.024455308090643, + "content": "the impacts of sea level rise and extreme\"\n\"weather events being felt most strongly in countries with very minimal carbon footprints, raising questions of equity\"\n\"and justice and how to address them [49, 126, 151, 177]. Similarly, the majority of climate-focused AI solutions over-\"\n\"look issues of justice and power, focusing predominantly on the climate-positive aspects of technologies and not wh", + "type": "table" + }, + { + "rank": 15, + "score": 2.9830716329716007, + "content": "ghtsaswellastolimitdamagetoenvironment,\nthere are no official provisions regarding sustainability in any of their texts, and it remains to be seen how existing\nstandards for environmental impacts in all of these jurisdi ctions will apply to AI systems. Similarly, sustainability\nconsiderationswerealsolackinginthe2023US ExecutiveOr derregarding AI[20],whichdidnotmentionAI’sgreen-\nhouse gas emissions n", + "type": "text" + }, + { + "rank": 16, + "score": 2.9423697240405247, + "content": "0\nBridging the Gap:\n\"Integrating Ethics and Environmental Sustainability in AI Research and Practice\n21\"\n\"[153]\nDavid Patterson, Joseph Gonzalez, Quoc Le, Chen Liang, Lluis-Miquel Munguia, Daniel Rothchild, David So, Maud Texier, and Jeff Dean. 2021.\"\nCarbon emissions and large neural network training. arXiv preprint arXiv:2104.10350 (2021).\n\"[154]\nGiada Pistilli, Carlos Muñoz Ferrandis, Yacine Jer", + "type": "table" + }, + { + "rank": 17, + "score": 2.876946545790257, + "content": "]\nZachary C Lipton. 2018. The mythos of model interpretability: In machine learning, the concept of interpretability is both important and slippery.\"\n\"Queue 16, 3 (2018), 31–57.\"\n\"[114]\nYinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov.\"\n2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint", + "type": "table" + }, + { + "rank": 18, + "score": 2.839070741870796, + "content": "hareunattainabletomany membersoftheAIcommunity,aswellasthepropagation\nof biases via their usage. Furthermore, the authors themsel ves note that there is currently no information available\nabout the embodied emissions linked to manufacturing GPUs, so it is impossible to estimate what portion of the\noverall carbonfootprintthis represents. This highlights that the emphasis onenvironmental sustainabil", + "type": "text" + }, + { + "rank": 19, + "score": 2.7311997228700937, + "content": "yment, proposing a holistic,\nlife cycle approach to estimating emissions [121]. There have also been proposals\"\n\"arguing for putting sustainability in the center of AI development and deployment [203], as well as frameworks for\"\n\"certifying the sustainability of AI systems [25], across all the different pillars of sustainability (i.e. social, environmen-\"\n\"tal and economic) [70]. However, we are st", + "type": "table" + }, + { + "rank": 20, + "score": 2.7083322592867085, + "content": "\"\n\"portant principles. However, while both EU AI Act [2022], as well as similar regulatory initiatives in China [2023] and\"\n\"Canada [2023] point to the need to protect both fundamental human rights as well as to limit damage to environment,\"\n\"there are no official provisions regarding sustainability in any of their texts, and it remains to be seen how existing\"\n\"standards for environmental\nimpacts i", + "type": "table" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/luccioni2025c_chunks_ranked.json b/wattbot2025/data/ranked/luccioni2025c_chunks_ranked.json new file mode 100644 index 00000000..8bc239ce --- /dev/null +++ b/wattbot2025/data/ranked/luccioni2025c_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.913278221908811, + "content": "hich was commissioned by\nGoogle and published ahead of COP2649. The reasoning behind the 5-10% reduction estimate is unclear and the underlying\ncalculations are not detailed beyond the explanation that they are based on BCG’s experience in dealing with their clients and\nusing AI to optimize and improve existing processes. The second, Google-commissioned BCG study provides slightly more\ndetail in t", + "type": "text" + }, + { + "rank": 2, + "score": 2.895039966300603, + "content": "g, C. The evolved transformer.\nIn Chaudhuri, K. & Salakhutdinov, R. (eds.) Proceedings of the\"\n\"36th International Conference on Machine Learning, vol. 97 of Proceedings of Machine Learning Research, 5877–5886\"\n\"(PMLR, 2019).\"\n\"32. Hao, K. Training a single ai model can emit as much carbon as five cars in their lifetimes. MIT technology Rev. 75, 103\"\n(2019).\n\"33. Toews, R. Deep learning’s carbon e", + "type": "table" + }, + { + "rank": 3, + "score": 2.7380728676026953, + "content": "2024).\n57.Schmidt, V . et al. Codecarbon: Estimate and track carbon emissions from machine learning computing (2021).\n58.Lannelongue, L., Grealey, J. & Inouye, M. Green algorithms: Quantifying the carbon footprint of computation. Adv. Sci.\n2100707 (2021).\n59.Mitchell, M. et al. Model cards for model reporting. In Proceedings of the conference on fairness, accountability, and\ntransparency , 220–229", + "type": "text" + }, + { + "rank": 4, + "score": 2.6669020553157825, + "content": "0,1\n\"56. Ambrose, J. & Hern, A. Ai will be help rather than hindrance in hitting climate targets, bill gates says (2024).\",\n\"57. Schmidt, V. et al. Codecarbon: Estimate and track carbon emissions from machine learning computing (2021).\",\n\"58. Lannelongue, L., Grealey, J. & Inouye, M. Green algorithms: Quantifying the carbon footprint of computation. Adv. Sci.\",\n2100707 (2021).,\n\"59. Mitchell, M. e", + "type": "table" + }, + { + "rank": 5, + "score": 2.6582703506262972, + "content": "er.ai/rankings?view=month (2025). Accessed: 2025-06-03.\n29.Lovins, A. B. Artificial intelligence meets natural stupidity: Managing the risks (2025).\n30.Vaswani, A. et al. Attention is all you need. Adv. neural information processing systems 30(2017).\n31.So, D., Le, Q. & Liang, C. The evolved transformer. In Chaudhuri, K. & Salakhutdinov, R. (eds.) Proceedings of the\n36th International Conference o", + "type": "text" + }, + { + "rank": 6, + "score": 2.641176854702522, + "content": "cation being picked\nup by numerous media outlets (including MIT Technology Review32and Forbes33). The “five cars” number has since been\nmisinterpreted as a proxy for the carbon footprint of training AI models at large, which is misleading given the diversity of\narchitectures, training approaches and electricity sources used for powering AI model training; the original article reports AI\ntraining w", + "type": "text" + }, + { + "rank": 7, + "score": 2.350894217960084, + "content": "A. Energy and policy considerations for deep learning in NLP. arXiv preprint\narXiv:1906.02243 (2019).\n13.Luccioni, A. S. & Hernandez-Garcia, A. Counting carbon: A survey of factors influencing the emissions of machine\nlearning. arXiv preprint arXiv:2302.08476 (2023).\n14.Dodge, J. et al. Measuring the carbon intensity of AI in cloud instances. In Proceedings of the 2022 ACM Conference on\nFairness, ", + "type": "text" + }, + { + "rank": 8, + "score": 2.2900839079104873, + "content": "versity of\"\n\"architectures, training approaches and electricity sources used for powering AI model training; the original article reports AI\"\n\"training workloads emitting as little as 26 pounds (11.8 kg) CO2e (assuming U.S. average energy carbon emissions intensity),\"\nand AI model training more broadly often requires even less energy and corresponding emissions.\n\"Further, the NAS training workload", + "type": "table" + }, + { + "rank": 9, + "score": 2.256756273124811, + "content": "Tech. Rep. ITU-T L.1480, ITU (2022). Accessed: 2025-06-01.\n51.WBCSD. Guidance on Avoided Emissions. Tech. Rep., WBCSD (2023). Accessed: 2025-06-01.\n52.Das, K. P. & Chandra, J. A survey on artificial intelligence for reducing the climate footprint in healthcare. Energy Nexus\n9, 100167 (2023).\n53.The Environment. Artificial intelligence can reduce 5 to 10 percent ghg emission: Study (2022).\n54.Kakka", + "type": "text" + }, + { + "rank": 10, + "score": 2.2096284655295766, + "content": "he current allowance for market-based\naccounting enables companies to significantly under-report their actual AI-related emissions through renewable energy\ncertificates, creating the same problematic disconnect from reality that has undermined carbon offsetting credibility56. For AI\nservices consuming substantial electricity across distributed data centers, mandatory location-based accounting woul", + "type": "text" + }, + { + "rank": 11, + "score": 2.1898874614539148, + "content": "er-report\ntheir actual AI-related emissions through renewable energy\"\n\"certificates, creating the same problematic disconnect from reality that has undermined carbon offsetting credibility 56. For AI\"\n\"services consuming substantial electricity across distributed data centers, mandatory location-based accounting would ensure\"\nenvironmental transparency frameworks capture the true systemic climate ", + "type": "table" + }, + { + "rank": 12, + "score": 2.1852714189885414, + "content": "(by sharing compute data like GPU type and training length, as well as by\nreleasing their model weights to enable efficiency analysis). In terms of token usage, 84% of LLM usage is through models\nwith no disclosure, 14% for indirectly disclosed models, and only 2% for models with direct disclosure. This indicates that the\nmajority of users who interact with LLMs have no information about their env", + "type": "text" + }, + { + "rank": 13, + "score": 2.052367312333017, + "content": "or energy after all. MIT Technology Review (online). Accessed:\n2025-06-01.\n46.Berry Zwets. Researchers claim to cut energy consumption AI 95 percent. Techzine (online). Accessed: 2025-06-01.\n47.Adam Clark Estes. Should you feel guilty about using AI? V ox (online). Accessed: 2025-06-01.\n48.Degot, C., Duranton, S., Frédeau, M. & Hutchinson, R. Reduce carbon and costs with the power of ai. Boston Co", + "type": "text" + }, + { + "rank": 14, + "score": 2.010286772936046, + "content": "about using AI? Vox (online). Accessed: 2025-06-01.\n\"48. Degot, C., Duranton, S., Frédeau, M. & Hutchinson, R. Reduce carbon and costs with the power of ai. Boston Consult.\"\nGroup 26 (2021).\n\"49. Dannouni, A. et al. Accelerating climate action with ai. Boston Consult. Group Special Rep. Google (2023).\"\n\"50.\nITU. Enabling the Net Zero transition: Assessing how the use of information and communicati", + "type": "table" + }, + { + "rank": 15, + "score": 2.002076890139238, + "content": "tle as 0.8 MWh (OLMo 20M) to 3,500 MWh\n(LLaMa 4 Scout), with associated GHG emissions varying even more significantly (due to variation in the carbon intensity of\nelectricity across training locations). Inference workloads also show wide variation depending on model size, architecture and\n3/12\n\ntask type, with GPU energy usage for 1,000 queries spanning from just 0.06 Wh (bert-tiny) to over 3,426 ", + "type": "text" + }, + { + "rank": 16, + "score": 1.977844715619471, + "content": "ssions (CO2e), or about five times the emissions of a car during its lifetime, including fuel.\"\n\"The research article was written for a specialized audience of AI and NLP researchers, who would have the background\"\n\"knowledge to understand the appropriate scoping for the estimate. However, an author’s tweet publicizing the paper and\"\n\"featuring a table containing the “five cars” estimate was widel", + "type": "table" + }, + { + "rank": 17, + "score": 1.5031966531900047, + "content": "0,1\n\"in environmental impact transparency: some models disclose sufficient details to enable impact estimation, whereas others\",\nprovide no information at all regarding their approach.,\n,\"Overall, we find that models exhibit three transparency categories:\"\n,• Direct Disclosure: Developers explicitly reported energy or GHG emissions. Note that this category includes methodolo-\n,\"gies ranging from e", + "type": "table" + }, + { + "rank": 18, + "score": 1.4587725372443987, + "content": "carbon data associated with manufacturing AI accelerators and data center infrastructure,\"\n\"significantly extending existing environmental\nimpact models. Building on many of\nthese approaches, Morrison et al.26\"\n\"performed a holistic evaluation of the energy, carbon, and water impacts of AI hardware manufacturing, model development,\"\n\"and training, enhancing the accuracy of these metrics through th", + "type": "table" + }, + { + "rank": 19, + "score": 1.4526396799828134, + "content": "hat training an AI model of the LLaMa 3.1 scale can produce air pollutants\nequivalent to more than 10,000 round trips by car between Los Angeles and New York City. In another significant advancement,\nGoogle’s recent TPU lifecycle assessment25offered the most comprehensive cradle-to-grave environmental analysis of AI\nhardware to date, integrating embodied carbon data associated with manufacturing A", + "type": "text" + }, + { + "rank": 20, + "score": 1.4526396799828134, + "content": "ding their approach.\nOverall, we find that models exhibit three transparency categories:\n•Direct Disclosure : Developers explicitly reported energy or GHG emissions. Note that this category includes methodolo-\ngies ranging from estimation (e.g., using hardware TDP, country average carbon intensity) to measurements (i.e., using\ntools like CodeCarbon).\n•Indirect Disclosure : Developers provided trai", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/morrison2025_chunks_ranked.json b/wattbot2025/data/ranked/morrison2025_chunks_ranked.json new file mode 100644 index 00000000..53462207 --- /dev/null +++ b/wattbot2025/data/ranked/morrison2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.119417831291344, + "content": "0\nPublished as a conference paper at ICLR 2025\nLuccioni et al. (2023) reported estimates for emissions from the manufacturing process (embodied\n\"emissions),\nfrom electricity consumption during training, and from electricity consumption of the\"\n\"cluster while it was idle (see their Table 2). Dodge et al.\n(2022) measured electricity consump-\"\ntion and carbon emissions for training language models an", + "type": "table" + }, + { + "rank": 2, + "score": 4.062932711058226, + "content": "granular\ntimesteps with region-specific carbon intensity, but did not measure development costs, water con-\nsumption, or inference. Similarly, developers of the Llama models (Touvron et al., 2023a;b; Dubey\net al., 2024) reported electricity consumption and carbon emissions estimates of training their final\nmodels; they did not estimate development cost or water consumption, and their approach to c", + "type": "text" + }, + { + "rank": 3, + "score": 3.8606137309868167, + "content": "missions estimates of training their final\"\n\"models; they did not estimate development cost or water consumption, and their approach to carbon\"\n\"intensity varied.5 Gemma developers (Gemma Team et al., 2024) only report a single number:\nthe\"\n\"total emissions from pretraining their models, not broken down by model or by different stages of\"\n\"training, or by electricity consumption and carbon intensi", + "type": "table" + }, + { + "rank": 4, + "score": 3.8487434231344544, + "content": "ail below.\n3.1 O PERATIONAL IMPACTS\nOperational environmental impacts of LLMs are those that arise directly from the development\nand use of models, and include the GHG emissions arising from energy sources used to power\nmodel training and deployment, including servers and data center cooling. We base our analysis of\noperational emissions around the following equation introduced by Schwartz et al. ", + "type": "text" + }, + { + "rank": 5, + "score": 3.7389149882158508, + "content": "ter usage, or embodied carbon, a few reports recently have included some estimates. For example,\n3https://ghgprotocol.org/sites/default/files/standards/ghg-protocol-revised.pdf\n4https://www.cnbc.com/2025/02/20/openai-tops-400-million-users-despite-deepseeks-emergence.\nhtml\n2\n\nPublished as a conference paper at ICLR 2025\nLuccioni et al. (2023) reported estimates for emissions from the manufacturing", + "type": "text" + }, + { + "rank": 6, + "score": 3.5895848027806183, + "content": "e, and use this to estimate the total carbon emissions and water consumption during each\nstage. We follow previous work (Luccioni et al., 2023; Dubey et al., 2024; Gemma Team et al.,\n2024) to calculate CO 2emissions (CO 2e) from power consumption:\nCO 2e=P·PUE·CI (2)\nwhere the total carbon emissions is equal to the power usage P, multiplied by the power usage\neffectiveness ( PUE )6of the data cente", + "type": "text" + }, + { + "rank": 7, + "score": 3.324019000181216, + "content": "0k 459 813 159 33 yrs, 1 mo 843 7 yrs, 5 mo\nHardware manufacturing NVIDIA does not release the embodied carbon emissions or water\nconsumption about the hardware it produces, so we assume the same embodied carbon emissions\nas Luccioni et al. (2023), or 3700 kg of CO 2eq per 8x server node, equal 463 kg per GPU. There\nis little public information on how much water is required to produce a single GPU", + "type": "text" + }, + { + "rank": 8, + "score": 3.288418438187392, + "content": "g each\"\n\"stage. We follow previous work (Luccioni et al., 2023; Dubey et al., 2024; Gemma Team et al.,\"\n2024) to calculate CO2 emissions (CO2e) from power consumption:\n\"(2)\nCO2e = P · P U E · CI\"\n\"where the total carbon emissions is equal\nto the power usage P, multiplied by the power usage\"\n\"effectiveness (PUE)6 of\nthe data center, multiplied by the carbon intensity CI of\nthe local power\"\n\"grid. W", + "type": "table" + }, + { + "rank": 9, + "score": 3.060912677580939, + "content": "4) documents electricity consumption per model, and uses region-specific carbon intensity to\nestimate emissions for two regions, but does not estimate other environmental impacts. The OLMo\n2 report (OLMo et al., 2025) again documents electricity consumption per model and uses region-\nand datacenter-specific intensity factors to estimate emissions and also water consumption, but does\nnot measure de", + "type": "text" + }, + { + "rank": 10, + "score": 3.0528471595067135, + "content": "pectively, and carbon intensity of 0.332 kg CO 2e / kWh. Note the difference in units for energy consumption\nand carbon emissions, namely MWh →kWh, tons →grams CO 2eq, and kL →L. The measurements reported\nin this table account for the GPU processes associated with active inference, but not CPU or RAM associated\nwith e.g. server overhead. Thus, these numbers can be considered as lower bounds on usa", + "type": "text" + }, + { + "rank": 11, + "score": 2.9276503825882934, + "content": "can be considered as lower bounds on usage in similar settings.\",,\n,Also of note is the relatively small variability in carbon emissions and water consumption across different model,,\n,\"sizes in cases where batches are not saturated, despite faster inference in smaller models when fully saturated;\",,\n,greater peak efficiency does not guarantee efficient deployment if inference is not optimized. We", + "type": "table" + }, + { + "rank": 12, + "score": 2.8474326658841513, + "content": "15, this is\nequivalent to 6.5 tanker trucks’ worth of gasoline burned, emissions from the average yearly energy\nuse for 98.2 homes in the U.S., or the amount of carbon sequestered by 472 acres of U.S. forests in\none year. We additionally estimate we consumed at least 2,769 kL of water, which is equivalent to\nabout 24 and a half years of water consumption by the average person in the U.S.16\nOther C", + "type": "text" + }, + { + "rank": 13, + "score": 2.798310522617105, + "content": "cope 2 CO2\"\nemissions in accordance with the Greenhouse\n\"OLMo 700M\nGas Protocol’s definitions,3 and Scope 1 and 2\n101\"\n\"water consumption following Li et al.\n(2023);\"\nOLMo 150M\n\"in addition, we calculate “upstream” embod-\"\n\"ied carbon and water consumption, and provide\"\n\"OLMo 20M\n100\"\n“downstream” estimates from use of our mod-\n\"100\n101\n102\"\n\"els (which are part, but not all, of Scope 3).\"\nCarbon ", + "type": "table" + }, + { + "rank": 14, + "score": 2.780203245717691, + "content": "okens. To do this, we calculate Scope 2 CO 2\nemissions in accordance with the Greenhouse\nGas Protocol’s definitions,3and Scope 1 and 2\nwater consumption following Li et al. (2023);\nin addition, we calculate “upstream” embod-\nied carbon and water consumption, and provide\n“downstream” estimates from use of our mod-\nels (which are part, but not all, of Scope 3).\nImportantly, we calculate (i) electric", + "type": "text" + }, + { + "rank": 15, + "score": 2.762334558024964, + "content": "a range of industries, such as transportation and end of life hardware disposal.\nWhile the costs we report above represent a large portion of the total development process, more\ntransparency is needed to understand the full impact of model training.\n4.2 S IMULATING DEPLOYMENT & INFERENCE\nWe report simulated inference costs; that is, we explore the question of what our models’ impact\nmight be if th", + "type": "text" + }, + { + "rank": 16, + "score": 2.7534881822144697, + "content": "used a region-specific carbon intensity. All 3\nassumed 100% GPU power draw throughout training.\n3\n\nPublished as a conference paper at ICLR 2025\nwhere the cost of a scientific result R(e.g. a claim that a particular training setup reaches Xaccuracy\non benchmark Y) is proportional to the product of the cost of processing a single example E, the\nsize of the training dataset D, and the number of hyper", + "type": "text" + }, + { + "rank": 17, + "score": 2.6962676211229004, + "content": "0,1,2,3,4,5,6\nAlso of note is the relatively small variability in carbon emissions and water consumption across different model,,,,,,\n\"sizes in cases where batches are not saturated, despite faster inference in smaller models when fully saturated;\",,,,,,\ngreater peak efficiency does not guarantee efficient deployment if inference is not optimized. We do not report,,,,,,\n”break-even” points for Qwe", + "type": "table" + }, + { + "rank": 18, + "score": 2.529660780464472, + "content": "Published as a conference paper at ICLR 2025\nHOLISTICALLY EVALUATING THE ENVIRONMENTAL\nIMPACT OF CREATING LANGUAGE MODELS\nJacob Morrison1Clara Na2Jared Fernandez2\nTim Dettmers1,2Emma Strubell1,2Jesse Dodge1\n1Allen Institute for AI2Carnegie Mellon University\njacobm@allenai.org\nABSTRACT\nAs the performance of artificial intelligence systems has dramatically increased,\nso too has the environmental imp", + "type": "text" + }, + { + "rank": 19, + "score": 2.4295757739023154, + "content": "0\nABSTRACT\n\"As the performance of artificial\nintelligence systems has dramatically increased,\"\nso too has the environmental impact of creating these systems. While many model\ndevelopers release estimates of the power consumption and carbon emissions from\n\"the final\ntraining runs for\ntheir latest models,\nthere is comparatively little trans-\"\n\"parency into the impact of model development, hardware m", + "type": "table" + }, + { + "rank": 20, + "score": 2.3282479650036745, + "content": "wer usage during development and\ntraining, we analyze detailed time series data for a single node throughout each run, logging power\ndata at sub-second intervals, and extrapolate to the total number of nodes. As we only measure GPU\npower consumption, our estimates should be viewed as a lower bound on the true amount of power\nconsumed during development and training.\n3.2 E MBODIED IMPACTS\nEmbodied ", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/patterson2021_chunks_ranked.json b/wattbot2025/data/ranked/patterson2021_chunks_ranked.json new file mode 100644 index 00000000..e6fad8f5 --- /dev/null +++ b/wattbot2025/data/ranked/patterson2021_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 6.413756155987723, + "content": "200 * 208 * 1.10 / 1000) * 0.431 / 1000 = 3.2 tCO 2 e (7096 lbs) . 36 \nThis actual emissions value is 88X smaller than the incorrect estimate of the carbon emissions of this \nsearch found in Strubell et al. If we reran the NAS search today on TPU v2s in Google’s Iowa datacenter \nwith 24/", + "type": "text" + }, + { + "rank": 2, + "score": 6.093471368084428, + "content": "shown both before (“gross”) and after (“net”),,,,,\n\"accounting for 24/7 reduction via real time, local carbon free energy purchases (Appendix B). To help\",,,,,\n\"put the CO 2 e numbers in perspective, a single passenger round trip SF-NY is ~1.2t CO 2 e (Table 2).\",,,,,", + "type": "table" + }, + { + "rank": 3, + "score": 4.519900202596487, + "content": "0,1,2,3,4,5\nGross CO 2 e for Model Training (metric ton) (Section,,,,,\n,0.1357,0.1055,0.0883,0.0189,0.0143\n2.4 and Appendix D),,,,,\nNet CO 2 e for Model Training (metric ton) (Section,,,,,\n,0.1357,0.0177,0.0148,0.0032,0.0024\n2.4 and Appendix D),,,,,\n% 24/7 net carbon free energy (CY 2019),N/A,,,78%,\nTable 1. See Appendix A for more ", + "type": "table" + }, + { + "rank": 4, + "score": 4.3503268452777535, + "content": "mer \nfor P100 and TPU v2 are based on power measurements. 5 Evolved Transformer (Medium) reached the \nsame accuracy as Transformer (Big) in [So19]. CO 2 e is shown both before (“gross”) and after (“net”) \naccounting for 24/7 reduction via real time, local carbon free energy purchases (Appendix B). To h", + "type": "text" + }, + { + "rank": 5, + "score": 4.120851408745134, + "content": "at the actual search used one TPU v2 chip to fit the same\"\nbatch size as one P100)\nTraining speed of Transformer Base on P100 from [Vas17]:\n\"hours_per_train_steps = 12 hours / 100,000 = 0.00012 (Section 5.2 in [Vas17])\"\n\"CO 2 e = 1 * 979,000,000 * 0.00012 * 0.2855296 = 33,544 lbs (15.2 t)\"\nAppendix ", + "type": "table" + }, + { + "rank": 6, + "score": 3.532613152675838, + "content": "reduce it, we \nendorse prior calls for new publication norms for computationally intensive ML models: \n1 Google \n2 University of California, Berkeley \n3 “CO 2 e” means CO 2 equivalent emissions , accounting for carbon dioxide and all the other greenhouse gases as well: \nmethane, nitrous oxide, ... (calculated ", + "type": "text" + }, + { + "rank": 7, + "score": 3.5245641716105722, + "content": "ion 4.1 second paragraph in [So19]). \nnum_chips = 1 (Section 4.3 in [So19], note that the actual search used one TPU v2 chip to fit the same \nbatch size as one P100) \nTraining speed of Transformer Base on P100 from [Vas17]: \nhours_per_train_steps = 12 hours / 100,000 = 0.00012 (Section 5.2 in ", + "type": "text" + }, + { + "rank": 8, + "score": 3.2094517449940367, + "content": "currently deploying numerous TPU v4s, many of which will be located in windy Oklahoma, \nwhose net CO 2 e/KWh is even lower than Iowa. \n●Fallacy: There is no business reason to reduce carbon emissions . Reducing climate change certainly \nhas long-term economic benefits for everyone. Google has been carbon ne", + "type": "text" + }, + { + "rank": 9, + "score": 3.13155089583522, + "content": "\"We calculate the energy use and carbon footprint of several recent large models— T5 , Meena , GShard ,\"\n\"Switch Transformer , and GPT-3 —and refine earlier estimates for the neural architecture search that found\"\nEvolved Transformer .\nWe highlight the following opportunities to improve energy efficiency and CO 2 eq", + "type": "table" + }, + { + "rank": 10, + "score": 3.1104942374672477, + "content": "rmation and radiative forcing of contrail cirrus. Nature communication s. 2018 May 8;9(1):1-7.,\n,https://www.nature.com/articles/s41467-018-04068-0 .\n\"[Kuc18] Kuczmarski, J. and Johnson, M., 2018. Gender-aware natural language\",\n,translation. www.tdcommons.org/dpubs_series/1577/ .\n\"[Lac19] Lacoste, A., Luccioni, A., Schmidt, V. and Dandr", + "type": "table" + }, + { + "rank": 11, + "score": 3.083790110009565, + "content": "Conventional carbon offsets try to create economic incentives to create projects that avoid or remove \nCO 2 e. When pursuing the mitigation of carbon emissions from electricity production and consumption, a \ncompany can match their MWh of consumption with MWh of clean energy through certificates called REC s ", + "type": "text" + }, + { + "rank": 12, + "score": 3.060139508309333, + "content": "r depth should take a look at [Ryo14, Goog16, Goo21].\"\nConventional carbon offsets try to create economic incentives to create projects that avoid or remove\n\"CO 2 e. When pursuing the mitigation of carbon emissions from electricity production and consumption, a\"\ncompany can match their MWh of consumption with M", + "type": "table" + }, + { + "rank": 13, + "score": 3.0399859134328278, + "content": "ni, A., Schmidt, V. and Dandres, T., 2019. Quantifying the carbon emissions of machine \nlearning. arXiv preprint arXiv:1910.09700 . \n[Lan20] Lannelongue, L., Grealey, J. and Inouye, M., 2020. Green algorithms: Quantifying the carbon footprint of \ncomputation. arXiv: 2007.07610 . \n[Leo19] Leopold, G. March 19, 201", + "type": "text" + }, + { + "rank": 14, + "score": 2.883425317947308, + "content": "0\nthat would be a great step forward. Perhaps ML practitioners could study the total lifecycle to develop rules of\nthumb to estimate the overall carbon footprint based on its final training cost. 16\nThe next subsection also emphasizes the value of measurement.\n\"Figure 5. Measured vs peak performance, measured ", + "type": "table" + }, + { + "rank": 15, + "score": 2.858922513427567, + "content": "ould reduce overall total carbon emissions if that model \nalso cut serving energy by 20%. Because energy usage during training is more isolated and thus easier to \ninvestigate than inference, we focus on it in this paper, but keep in mind that the carbon footprint of inference is \nsignificant. \nAn M", + "type": "text" + }, + { + "rank": 16, + "score": 2.7343902734800483, + "content": "on Computer Architecture. \n[Kap20] Kaplan, J., McCandlish, S., Henighan, T., Brown, T.B., Chess, B., Child, R., Gray, S., Radford, A., Wu, J. and \nAmodei, D., 2020. Scaling laws for neural language models. arXiv preprint arXiv:2001.08361. \n[Kär18] Kärcher B. Formation and radiative forcing of contrail cirrus. ", + "type": "text" + }, + { + "rank": 17, + "score": 2.636851091000899, + "content": "cy  \nThere are many algorithmic techniques that can improve the energy efficiency of machine learning models. \nSome techniques can achieve the same accuracy with less overall computation. Others can use a large, \nalready-trained model as a starting point and yield a lighter-weight, more computationally efficient", + "type": "text" + }, + { + "rank": 18, + "score": 2.5460309262562344, + "content": "occurred in the absence of a market \nfor offset credits. Additionality is essential for the quality of carbon offset credits—if their associated \nCO 2 e reductions are not additional, then purchasing offset credits in lieu of reducing your own \nemissions will make climate change worse. \n●The Grid : The tran", + "type": "text" + }, + { + "rank": 19, + "score": 2.5460309262562344, + "content": "to increase,,,,,,,,\naccuracy while lowering energy consumption and CO 2 e that could bend the curve of ML carbon footprint,,,,,,,,\ngrowth for computationally intensive NLP models.,,,,,,,,\n,The following sections summarize the findings that led to these recommendations. They also document our,,,,,,,\n\"CO 2 e estimates, highlig", + "type": "table" + }, + { + "rank": 20, + "score": 2.52122020355897, + "content": "eb services claimed that 90% of the ML demand in the cloud is for inference [Bar19]. Given its,,,,,,,,\n\"substantial role in the ML model lifecycle, Alibaba, Amazon, Google, and NVIDIA designed ML accelerators\",,,,,,,,\n\"solely for inference. If the total ML energy is split 10% on training and 90% on serv", + "type": "table" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/rubei2025_chunks_ranked.json b/wattbot2025/data/ranked/rubei2025_chunks_ranked.json new file mode 100644 index 00000000..876ce5ed --- /dev/null +++ b/wattbot2025/data/ranked/rubei2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.431007954157246, + "content": "n last for weeks or even months. Therefore, measuring\nthe energy consumption in terms of carbon emissions is\nparticularly challenging in those environments due to several\nfactors, e.g., parallel jobs or the non-exclusive use of the\ncluster.\nMoreover, even well-maintained LLMs leaderboard bench-\nmarks [19]–[21] do not report energy consumption, focusing\ninstead on accuracy metrics. Figure 1 shows t", + "type": "text" + }, + { + "rank": 2, + "score": 4.394225251166906, + "content": "eveloped to\",related tasks.\nmeasure the carbon emissions associated with code execution.,While developing a comprehensive methodology for mea-\n\"Among these,\nthe CodeCarbon tool\n[16]\nis a widely adopted\",\"suring LLM energy consumption is beyond this paper’s scope,\"\nPython library that estimates the energy consumption of code,we focus on reducing these emissions through efficient PETs.\n\"executions.\n", + "type": "table" + }, + { + "rank": 3, + "score": 4.25849883329765, + "content": "s the trade-offs between energy\nconsumption in terms of carbon emission, execution time,\nand generated code accuracy to investigate the balance\nbetween energy efficiency and model accuracy;\n•We provide a replication package1to foster further re-\nsearch on the topic.\n1https://github.com/riccardoRubei/Greens-2025-Replication-PackagearXiv:2501.05899v1 [cs.SE] 10 Jan 2025\n\nFig. 1: Carbon emissions o", + "type": "text" + }, + { + "rank": 4, + "score": 4.060532137056889, + "content": "e exact matches\nrise from 63 to 82, reflecting a 23% increase. Both one-\nshot and few-shots see substantial gains with C3, achieving\napproximately a 44% improvement. Interestingly, with C4,\nzero-shot fails to achieve any exact matches.\nFigure 4b shows the impact of custom tags on edit distance\nmetrics, where an edit distance of 0 indicates a perfect result.\nOverall, custom tags contributed to a re", + "type": "text" + }, + { + "rank": 5, + "score": 3.735445279120476, + "content": "cially when consider-\ning the growing scope of LLM-based implementations and\ntheir integration into everyday life. This highlights the need\nto reduce the carbon footprint of LLMs and to examine the\ndetails that contribute to the reported figures.\nTo address the environmental impact of software, a range of\nenergy monitoring tools [5], [6] has been recently developed to\nmeasure the carbon emissions ", + "type": "text" + }, + { + "rank": 6, + "score": 3.4375860542900334, + "content": "memory requirements of LLMs by lowering the\n\"ing the growing scope of LLM-based implementations\nand\",\"precision of\ntheir numerical\nrepresentations (e.g.,\nfrom 32-bit\"\n\"their\nintegration into everyday life.\nThis highlights the need\",\"to 8-bit). This compression speeds up inference, making LLMs\"\nto reduce the carbon footprint of LLMs and to examine the,\"more efficient with minimal\nimpact on accuracy", + "type": "table" + }, + { + "rank": 7, + "score": 3.3869855904651747, + "content": "s —LLMs, Generative AI, Prompt Engineering,\nEnergy Consumption.\nI. I NTRODUCTION\nThe environmental impact of software systems has been\na growing concern in recent years [1], [2], thus fostering\nthe development of green software engineering (GSE) [3] by\nproposing dedicated methodologies [4], frameworks [5], [6],\nand guidelines [7]. Nevertheless, the rise of AI-intensive sys-\ntems has posed new chal", + "type": "text" + }, + { + "rank": 8, + "score": 2.9799424648840027, + "content": "e\nlearning benchmark dataset for code understanding and generation,”\ninThirty-fifth Conference on Neural Information Processing Systems\nDatasets and Benchmarks Track (Round 1) .\n[15] A. Dubey, A. Jauhri, A. Pandey et al. , “The llama 3 herd of models,”\n2024. [Online]. Available: https://arxiv.org/abs/2407.21783\n[16] M. C. Impact, “Codecarbon: A tool to estimate the carbon emissions\nof machine lear", + "type": "text" + }, + { + "rank": 9, + "score": 2.9799424648840027, + "content": "0,1,2,3,4,5,6,7\n\"footprint\n[12]. Moreover,\nassessing them is\nchallenging due\",,can play a key role,,in reducing the,,energy consumption of,\n\"i)\nii)\nto\nhigher\nvariability\nin\nthe\ngenerated\ncode\nand\nthe\",,LLMs without compromising their performance.,,,,,\n\"lack of\nstandardized guidelines and information for measur-\",,The main contributions of,,this work are as follows:,,,\n\"ing\ncarbon\nemissions\neven\nin", + "type": "table" + }, + { + "rank": 10, + "score": 2.941792277781013, + "content": "s-\",for evaluating code completion when using LLMs [17].\ntems has posed new challenges regarding energy consumption,\"Our findings reveal\nthat\nthe energy consumption of LLMs\"\nand carbon emissions [8].,for the inference phase can be reduced by using the introduced\n\"In\nparticular,\nboth\ntraining\nand\nquerying\nlarge\nlanguage\",\"custom tags. Moreover, we show that\nthe energy consumption\"\n\"models (LLMs)\nto", + "type": "table" + }, + { + "rank": 11, + "score": 2.929291697912831, + "content": "measuring\",\"most basic PET is\nzero-shot,\nin which the LLM is given a\"\n\"the\nenergy\nconsumption\nin\nterms\nof\ncarbon\nemissions\nis\",\"query without\nany example of outputs, which are\nexpected\"\nparticularly challenging in those environments due to several,\"from the given inputs\n[23].\nIn contrast, one-shot prompting\"\n\"factors,\ne.g.,\nparallel\njobs\nor\nthe\nnon-exclusive\nuse\nof\nthe\",\"provides the model with a ", + "type": "table" + }, + { + "rank": 12, + "score": 2.880333998930991, + "content": "0,1\nFig. 1: Carbon emissions of GPT-3 models as reported in [18].,\nII. BACKGROUND,\"it\ncan\nestimate\nthe\ncarbon\nintensity\nof\nthe\nregion where\"\n,\"the\ncomputing\nis\ndone. This\nstudy\nfocuses\non\nthe\nenergy\"\n\"While measuring traditional\nsoftware\nimpact\nin terms of\",\"consumption related to GPU usage without\nconsidering the\"\n\"emissions\nis well-established [1],\n[7],\nassessing LLMs\ncon-\",carbon emission.\n\"sum", + "type": "table" + }, + { + "rank": 13, + "score": 2.8564637418098386, + "content": "0,1\n\"[39] conduct a controlled experiment\nin which code generated\",\"task that we decided to study,\nthus\ncode generation or\ntext\"\nby CodeLlama is compared with the human one considering,\"summarization might\nrequire different energy resources. We\"\n\"different\nlanguages,\ni.e., C++,\nJava, and Python,\ntested on a\",mitigated this threat focusing on the effects of the customiza-\ndedicated platform. The re", + "type": "table" + }, + { + "rank": 14, + "score": 2.8446763664016963, + "content": ". Finally, the measurements calculated on the inference\nwithout any customization are strictly related to the particulartask that we decided to study, thus code generation or text\nsummarization might require different energy resources. We\nmitigated this threat focusing on the effects of the customiza-\ntion.\nVII. C ONCLUSION AND FUTURE WORK\nMotivated by the increasing carbon emissions of LLMs, we\np", + "type": "text" + }, + { + "rank": 15, + "score": 2.8446763664016963, + "content": "particular, with the best configuration, zero-shot reduced\nthe consumption of about 7%, whereas one-shot and few-\nshots decreased their consumption of about 99% and 83%,\nrespectively. For future work, we plan to extend the study\nto additional LLMs and code-related tasks. In addition, we\nwill investigate advanced techniques, e.g., retrieval augmented\ngeneration (RAG) or fine-tuning, to further redu", + "type": "text" + }, + { + "rank": 16, + "score": 2.775945690453555, + "content": "n\nfootprint [12]. Moreover, assessing them is challenging due\ntoi)higher variability in the generated code and ii)the\nlack of standardized guidelines and information for measur-\ning carbon emissions even in dedicated model repositories\n[12]. While a plethora of approaches have been proposed to\nmeasure the impact on the hardware [13], we focus on the\nusage of prompt engineering techniques (PETs) to", + "type": "text" + }, + { + "rank": 17, + "score": 1.8284110047839826, + "content": "comprehensive methodology for mea-\nsuring LLM energy consumption is beyond this paper’s scope,\nwe focus on reducing these emissions through efficient PETs.\nBy utilizing custom tags, we aim to lower energy consumption\nin LLMs used for code-related tasks, offering an approach that\nbalances sustainability with performance.\n\nCodeXGlue\nDataset1\nLlama 3PET Selector Prompt Augmenter2 3\n 4\nCode Carbon \nEn", + "type": "text" + }, + { + "rank": 18, + "score": 1.7770977127568388, + "content": "hows that the proposed approach succeed in reducing the\ncarbon emission even though the region may impact the ob-\ntained results. Liu and Yin [37] investigate how to reduce and\nmeasure the consumption of pre-trained models by combining\nfine-tuning and efficient tokenizers. In particular, BERT, Distil-\nBERT, and T5 models are compared using SQuAD benchmark[38] in terms of accuracy and carbon emissi", + "type": "text" + }, + { + "rank": 19, + "score": 1.6333249120542226, + "content": "Prompt engineering and its implications on the\nenergy consumption of Large Language Models\nRiccardo Rubei\nUniversity of L’Aquila\nL’Aquila, Italy\nriccardo.rubei@univaq.itAicha Moussaid\nUniversity of L’Aquila\nL’Aquila, Italy\naicha.moussaid@student.univaq.itClaudio Di Sipio\nUniversity of L’Aquila\nL’Aquila, Italy\nclaudio.disipio@univaq.itDavide Di Ruscio\nUniversity of L’Aquila\nL’Aquila, Italy\ndavide.d", + "type": "text" + }, + { + "rank": 20, + "score": 1.5491216918770876, + "content": ".\nFig. 3: Energy consumption with different prompt configurations.\nC0 C1 C2 C3 C4020406080100120140Exatch Match Absolute NumberPET\nZeroShot\nOneShot\nFewShot\n(a) Exact Match.\nC0 C1 C2 C3 C4020406080100120Edit DistancePET\nZeroShot\nOneShot\nFewShot (b) Edit Distance.\nFig. 4: LLMs accuracy with different prompt configurations.\nV. R ELATED WORK\nAssessing LLMs energy consumption: Jagannadharao et al.\n[36]", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/samsi2024_chunks_ranked.json b/wattbot2025/data/ranked/samsi2024_chunks_ranked.json new file mode 100644 index 00000000..6e80bb43 --- /dev/null +++ b/wattbot2025/data/ranked/samsi2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 5.761047993181931, + "content": "f 250W. For\na 30% reduction in power from 250W to 175W, the inference\ntime increases by an average of 6.7% for a corresponding aver-\nage reduction in total energy by 23.21%. However, a reduction\nin power cap to 150W results in a much more significant\n(19.49%) increase in average inference time. These results\nshow that power capping as an energy savings intervention\ncan be effective when applied ap", + "type": "text" + }, + { + "rank": 2, + "score": 4.9241076879065035, + "content": "model performance\n\"Table III shows the relative change in total\ninference time,\",\n,\"at 250W to stay consistent with the settings in the rest of\nthe\"\n\"energy and token rate under power\ncap conditions. Results\",\n,experiments described here.\nshown here are calculated relative to a power cap of 250W. For,\n\"a 30% reduction in power from 250W to 175W,\nthe inference\",\ntime increases by an average of 6.7%", + "type": "table" + }, + { + "rank": 3, + "score": 4.577956452864955, + "content": "laws [2], [3]\nto safety concerns arising from the fact that these models are\ncapable of hallucinating or fabricating information, concerns\nabout these models in the educational and medical domain [4],\n[5], their carbon footprint, and many more.\nIn this paper, we focus primarily on understanding the\nsignificant amount of resources—time, computation, and\nenergy—required for using and deploying some ", + "type": "text" + }, + { + "rank": 4, + "score": 3.3209927267222894, + "content": "ecially the compute and energy costs required for\",\"cal concerns ranging from violations of copyright laws [2], [3]\"\n\"inference.\nInference\nenergy costs already receive\nless attention\",\"to safety concerns arising from the fact\nthat\nthese models are\"\nthan the energy costs of training LLMs—despite how often these,\n,\"capable of hallucinating or\nfabricating information, concerns\"\n\"large models are call", + "type": "table" + }, + { + "rank": 5, + "score": 3.240130965964572, + "content": "amically partition GPU resources. This opens up the\npotential to optimally partition high-end GPUs such as the\nA100s or H100s to co-locate multiple LLMs for inference—\nwith the potential of only minimal degradation to computa-\ntional performance.\nFinally, as AI compute requirements have increased, there\nis an increasing focus on approaches to reduce the carbon\nand energy footprints of datacenters ", + "type": "text" + }, + { + "rank": 6, + "score": 3.1756943506822815, + "content": "inference\nenergy\ncosts\nof\ndifferent\nsizes\nof\",\nLLaMA—a recent state-of-the-art LLM—developed by Meta AI,\"discuss the carbon footprint of language models such as BERT,\"\n\"on two generations of popular GPUs\n(NVIDIA V100 & A100)\",\"ELMo,\nand precursors\nto larger models\nsuch as GPT-3 and\"\n\"and two datasets\n(Alpaca\nand GSM8K)\nto\nreflect\nthe diverse\",\n,GPT-4 that power some of the popular AI chatbots toda", + "type": "table" + }, + { + "rank": 7, + "score": 3.1525023355970014, + "content": "ngly important\nconcern. In prior work, we have shown [25] that power capping\nGPUs during training of language models such as BERT [26]\nis an effective way of reducing the energy consumed training\nthese models. While the work in [25] focused on model\ntraining, in this paper, we focus on inference. In order to study\nthe effect of power capping on inference using large language\nmodels, we ran a limit", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "From Words to Watts: Benchmarking the Energy\nCosts of Large Language Model Inference\nSiddharth Samsi∗§, Dan Zhao†, Joseph McDonald∗, Baolin Li‡, Adam Michaleas∗,\nMichael Jones∗, William Bergeron∗, Jeremy Kepner∗, Devesh Tiwari‡, Vijay Gadepally∗\n∗MIT,†NYU,‡Northeastern University\nAbstract —Large language models (LLMs) have exploded in\npopularity due to their new generative capabilities that go far", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "receive less attention\nthan the energy costs of training LLMs—despite how often these\nlarge models are called on to conduct inference in reality (e.g.,\nChatGPT). As these state-of-the-art LLMs see increasing usage\nand deployment in various domains, a better understanding\nof their resource utilization is crucial for cost-savings, scaling\nperformance, efficient hardware usage, and optimal inference\n", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "eloped by Meta AI\non two generations of popular GPUs (NVIDIA V100 & A100)\nand two datasets (Alpaca and GSM8K) to reflect the diverse\nset of tasks/benchmarks for LLMs in research and practice.\nWe present the results of multi-node, multi-GPU inference using\nmodel sharding across up to 32 GPUs. To our knowledge, our\nwork is the one of the first to study LLM inference performance\nfrom the perspective ", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "ng text, images, and audio from which it’s\ntrained on. While GenAI is not entirely new, the recent\napplication and broad availability of this technology via tools\nsuch as Stable Diffusion [1], OpenAI’s ChatGPT, Google’s\nBard and integration into the Microsoft Bing search engine\nhas captured the imagination of the world and led to a massive\nsurge in interest in deploying these types of models acros", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "tive Agreement Number FA8750-19-2-1000. Any opinions, findings,\nconclusions or recommendations expressed in this material are those of the\nauthor(s) and do not necessarily reflect the views of the Assistant Secretary\nof Defense for Research and Engineering, or the United States Air Force.\nThe U.S. Government is authorized to reproduce and distribute reprints for\nGovernment purposes notwithstanding", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "3 and\nGPT-4 that power some of the popular AI chatbots today. Oth-\ners have also looked to larger language models; for instance,\nthe largest NVIDIA Megatron-LM model required 3,072 A100\nGPUs [7]–[9] for its training. While the complete details (time\nand resources used) of compute required for training GPT-\n3/4 are not available, several estimates for training [10], [11]\nand inference are publicly ", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "re of energy costs and their likely larger impact\non the environment [13]—especially since model inference\ncalls can occur more frequently than training/fine-tuning for\nreal-world deployments and applications.\nWe present the results of our inference experiments on\nLLaMA [14]: an open sourced pre-trained large language\nmodels by Meta AI. The LLaMA model is available in a\nnumber of sizes but, in mos", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "ngle node instances using smaller\nvariants of the model as a baseline comparison. We hope our\nwork will help illustrate some of the compute performance\nand energy utilization characteristics of LLM inference. We\nalso hope that our experiments, analysis, and data on real-arXiv:2310.03003v1 [cs.CL] 4 Oct 2023\n\nworld hardware will spur further analysis, benchmarking,\nand more open dissemination of ", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "wth in both\nthe speed of development as well as complexity of ever larger\nmodels. Over the past several years, competition has been\nfierce and the pace un-relenting as AI research groups across\nprivate companies and academic institutions have developed\nnew models whose performance continues to improve on a\nwide suite of natural language benchmarks but still requires\nsignificant amounts of compute ", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "te encoder-\ntype language models, green indicates encoder-decoder hybrid\nmodels, and the dark grey indicates decoder-style models. The\nbar-plot on the bottom right tallies the number of open/closed\nsource models developed by different companies/institutions.\nWe study LLaMA (outlined by the red arrow and red circle in\nthe diagram above) as an example of one of the more recent,\nmodern, and state-of-", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "h their own respective training setup,\narchitectural modifications, purposes or use-cases, etc. Large\nlanguage models and foundation models are best known for\ntheir sheer size, resource intensity (i.e., the amount of com-\nputational resources required for training/inference), and theirimpressive capabilities in tasks that include, but may not be\nlimited to, natural language.\nTypically, LLMs refer ", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "er-decoder architecture. Large language\nmodels can be considered a subset of large foundation models;\nwhereas LLMs focus almost exclusively on language data\nfor their inputs and outputs, large foundation models include\nmodels that allow for multiple modalities such as image and\ntext (e.g., GPT-4) or other modalities such as image generation\n(e.g., Stable Diffusion) or video generation (e.g., MidJo", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "e originally introduced in [16]. Most notably, the\nperformance of LLaMA rivaled or exceeded that of GPT-3 on\nmany NLP benchmarks and remains competitive with other\nstate-of-the-art LLMs [14]. Like other LLMs, LLaMA was\npre-trained on a large collection of data including but not\nlimited to CommonCrawl, Github, Wikipedia, etc. As of spring\n2023, alongside other recently timed releases of state-of-th", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/schwartz2019_chunks_ranked.json b/wattbot2025/data/ranked/schwartz2019_chunks_ranked.json new file mode 100644 index 00000000..f965cd58 --- /dev/null +++ b/wattbot2025/data/ranked/schwartz2019_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 5.879010288596595, + "content": "ublicly is a green success, and we would like\nto encourage organizations to continue to release their models in order to save others the costs of retraining them.\n4 Related Work\nRecent work has analyzed the carbon emissions of training deep NLP models [40] and concluded that computationally\nexpensive experiments can have a large environmental and economic impact. With modern experiments using such", + "type": "text" + }, + { + "rank": 2, + "score": 3.8263692714761754, + "content": "hyperparameter tuning. Failure to fully\nreport these experiments prevents future researchers from understanding how much effort is required to reproduce a\nresult or extend it [9].\nOur focus is on improving efficiency in the machine learning community, but machine learning can also be used\nas a tool for work in areas like climate change. For example, machine learning has been used for reducing emiss", + "type": "text" + }, + { + "rank": 3, + "score": 3.8259028226505656, + "content": "long-term trends, and are not isolated within NLP,\nbut hold true across machine learning.\nWhile some companies offset electricity usage by purchasing carbon credits, it is not clear that buying credits is\nas effective as using less energy. In addition, purchasing carbon credits is voluntary; Google cloud20and Microsoft\nAzure21purchase carbon credits to offset their spent energy, but Amazon’s AWS22", + "type": "text" + }, + { + "rank": 4, + "score": 3.3546057537770846, + "content": "ithub.com/sovrasov/flops-counter.pytorch\n20https://cloud.google.com/sustainability/\n21https://www.microsoft.com/en-us/environment/carbon\n22https://aws.amazon.com/about-aws/sustainability/\n23https://tinyurl.com/y2kob969\n8\n\n5 Conclusion\nThe vision of Green AI raises many exciting research directions that help to overcome the inclusiveness challenges of\nRed AI. Progress will reduce the computational ", + "type": "text" + }, + { + "rank": 5, + "score": 3.148556047487315, + "content": "green.\nWhen reporting the amount of work done by a model, we want to measure a quantity that allows for a fair com-\nparison between different models. As a result, this measure should ideally be stable across different labs, at different\ntimes, and using different hardware.\nCarbon emission Carbon emission is appealing as it is a quantity we want to directly minimize. Nonetheless it\nis impractical t", + "type": "text" + }, + { + "rank": 6, + "score": 3.1061431120739726, + "content": "0\nclearly underperforming can lead to great savings [21].\nReferences\n\"[1] Prabal Acharyya, Sean D Rosario, Roey Flor, Ritvik Joshi, Dian Li, Roberto Linares, and Hongbao Zhang.\"\n\"Autopilot of cement plants for reduction of fuel consumption and emissions, 2019.\nICML Workshop on Climate\"\nChange.\n\"[2] Dario Amodei and Danny Hernandez. AI and compute, 2018. Blog post.\"\n\"[3]\nJames S. Bergstra, R´emi Ba", + "type": "table" + }, + { + "rank": 7, + "score": 3.0584530573949262, + "content": "While many hyperparameter opti-\nmization algorithms exist which can reduce the computational expense required to reach a given level of performance\n[3, 10], simple improvements here can have a large impact. For example, stopping training early for models which are\nclearly underperforming can lead to great savings [21].\nReferences\n[1] Prabal Acharyya, Sean D Rosario, Roey Flor, Ritvik Joshi, Dian L", + "type": "text" + }, + { + "rank": 8, + "score": 2.98960194128209, + "content": "0\n\"5\nConclusion\"\nThe vision of Green AI raises many exciting research directions that help to overcome the inclusiveness challenges of\n\"Red AI. Progress will reduce the computational expense with a minimal reduction in performance, or even improve\"\n\"performance as more efficient methods are discovered. Also,\nit would seem that Green AI could be moving us in a\"\nmore cognitively plausible direction a", + "type": "table" + }, + { + "rank": 9, + "score": 2.313901296114107, + "content": "Green AI\nRoy Schwartz\u0003}Jesse Dodge\u0003}|Noah A. Smith}~Oren Etzioni}\n}Allen Institute for AI, Seattle, Washington, USA\n|Carnegie Mellon University, Pittsburgh, Pennsylvania, USA\n~University of Washington, Seattle, Washington, USA\nJuly 2019\nAbstract\nThe computations required for deep learning research have been doubling every few months, resulting in an\nestimated 300,000x increase from 2012 to 2018 [2", + "type": "text" + }, + { + "rank": 10, + "score": 2.2771037139363703, + "content": "0\nAbstract\n\"The computations required for deep learning research have been doubling every few months,\nresulting in an\"\n\"estimated 300,000x increase from 2012 to 2018 [2]. These computations have a surprisingly large carbon footprint\"\n\"[40]. Ironically, deep learning was inspired by the human brain, which is remarkably energy efficient. Moreover, the\"\n\"financial cost of the computations can make it d", + "type": "table" + }, + { + "rank": 11, + "score": 2.2154478987381876, + "content": "able progress on a broad range of capabilities in-\ncluding object recognition, game playing, machine translation, and more [36]. This progress has been achieved by\nincreasingly large and computationally-intensive deep learning models.1Figure 1 reproduced from [2] plots training\ncost increase over time for state-of-the-art deep learning models starting with AlexNet in 2012 [20] to AlphaZero in\n2017", + "type": "text" + }, + { + "rank": 12, + "score": 2.2154478987381876, + "content": "training or executing a model, and accordingly—\"\n\"generating an AI\nresult, as this amount depends highly on the local electricity infrastructure. As a result,\nit\nis not\"\ncomparable between researchers in different locations or even the same location at different times.\n\"Electricity usage\nElectricity usage is correlated with carbon emission while being time- and location-agnostic.\"\n\"Moreover, GPUs ", + "type": "table" + }, + { + "rank": 13, + "score": 2.198440499591202, + "content": "arper trend can be observed in NLP word embedding approaches by looking at ELMo [29] followed by BERT [8],\"\n\"openGPT-2 [30], and XLNet [48]. An important paper [40] has estimated the carbon footprint of several NLP models\"\n\"and argued that\nthis trend is both environmentally unfriendly (which we refer to as Red AI) and expensive,\nraising\"\nbarriers to participation in NLP research.\n\"This trend is dr", + "type": "table" + }, + { + "rank": 14, + "score": 2.181692233629117, + "content": "ifferent times.\nElectricity usage Electricity usage is correlated with carbon emission while being time- and location-agnostic.\nMoreover, GPUs often report the amount of electricity each of their cores consume at each time point, which facilitates\nthe estimation of the total amount of electricity consumed by generating an AI result. Nonetheless, this measure is\nhardware dependent, and as a result ", + "type": "text" + }, + { + "rank": 15, + "score": 2.1016382228016464, + "content": "0\n\"the cost of an experiment decomposes into the cost of a processing a single example,\nthe size of the dataset, and the\"\n\"number of experiments (Equation 1), reducing the amount of work in each of these steps will result in AI that is more\"\ngreen.\n\"When reporting the amount of work done by a model, we want\nto measure a quantity that allows for a fair com-\"\n\"parison between different models. As a ", + "type": "table" + }, + { + "rank": 16, + "score": 0.0, + "content": "om emerging economies, to engage in deep learning research.\nThis position paper advocates a practical solution by making efficiency an evaluation criterion for research along-\nside accuracy and related measures. In addition, we propose reporting the financial cost or “price tag” of developing,\ntraining, and running models to provide baselines for the investigation of increasingly efficient methods. O", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "footprint of several NLP models\nand argued that this trend is both environmentally unfriendly (which we refer to as Red AI) and expensive, raising\nbarriers to participation in NLP research.\nThis trend is driven by the strong focus of the AI community on obtaining “state-of-the-art” results,2as exemplified\nby the rising popularity of leaderboards [46, 45], which typically report accuracy measures bu", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "ctivity in Green AI—AI research that is more environmentally friendly and\ninclusive. We emphasize that Red AI research has been yielding valuable contributions to the field of AI, but it’s been\noverly dominant. We want to shift the balance towards the Green AI option —to ensure that any inspired undergraduate\nwith a laptop has the opportunity to write high-quality papers that could be accepted at p", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "me benchmark is greater than any previously reported system’s accuracy.\n1arXiv:1907.10597v3 [cs.CY] 13 Aug 2019\n\nFigure 1: The amount of compute used to train deep learning models has increased 300,000x in 6 years. Figure taken\nfrom [2].\nSpecifically, we propose making efficiency a more common evaluation criterion for AI papers alongside accuracy and\nrelated measures.\nAI research can be computatio", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "tational price\ntag of finding, training, and running models is a key Green AI practice (see Equation 1). In addition to providing\ntransparency, price tags are baselines that other researchers could improve on.\nOur empirical analysis in Figure 2 suggests that the AI research community has paid relatively little attention to\ncomputational efficiency. In fact, as Figure 1 illustrates, the computational", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/shen2024_chunks_ranked.json b/wattbot2025/data/ranked/shen2024_chunks_ranked.json new file mode 100644 index 00000000..15a34d2b --- /dev/null +++ b/wattbot2025/data/ranked/shen2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 0.0, + "content": "JetMoE\nJetMoE: Reaching Llama2 Performance with 0.1M Dollars\nYikang Shen∗\nMIT-IBM Watson AI Lab\nyikang.shn@gmail.comZhen Guo∗\nMIT EECS\nzguo0525@mit.edu\nTianle Cai\nPrinceton University\ntianle.cai@princeton.eduZengyi Qin\nMyShell.ai & MIT\nqinzy@mit.edu\nAbstract\nLarge Language Models (LLMs) have achieved remarkable results, but\ntheir increasing resource demand has become a major obstacle to the devel-", + "type": "text" + }, + { + "rank": 2, + "score": 0.0, + "content": ", with JetMoE-8B outperforming the Llama2-7B model and\nJetMoE-8B-Chat surpassing the Llama2-13B-Chat model. These results\nsuggest that LLM training can be much more cost-effective than gener-\nally thought. JetMoE-8B is based on an efficient Sparsely-gated Mixture-\nof-Experts (SMoE) architecture, composed of attention and feedforward\nexperts. Both layers are sparsely activated, allowing JetMoE-8B t", + "type": "text" + }, + { + "rank": 3, + "score": 0.0, + "content": "in this report to facilitate future efforts in the development of open\nfoundation models. This transparency aims to encourage collaboration and\nfurther advancements in the field of accessible and efficient LLMs. The mod-\nels are publicly available at https://github.com/myshell-ai/JetMoE .\n1 Introduction\nLarge Language Models (LLMs) have achieved remarkable results, but their increasing\nresource de", + "type": "text" + }, + { + "rank": 4, + "score": 0.0, + "content": "et al. 2023) use all of their parameters\nduring inference and training, which are referred to as dense models. Considering the\nsubstantial costs, the Mixture-of-Experts (MoE) architecture (Yuksel et al., 2012; Shazeer\net al., 2017; Du et al., 2022; Pan et al., 2024) has emerged as a popular solution, enabling\nparameter scaling while keeping computational costs modest. Recent applications of MoE\nar", + "type": "text" + }, + { + "rank": 5, + "score": 0.0, + "content": ". However, even though these models achieve excellent\nperformance, they are not truly open-sourced as the training recipes are not published and\nmay contain proprietary datasets inaccessible outside of large corporations. The open-source\ncommunity has also attempted to train MoE models, such as OpenMoE (Xue et al., 2024), but\nits performance is only on par with weak dense models with similar activ", + "type": "text" + }, + { + "rank": 6, + "score": 0.0, + "content": "an innovative MoE architecture inspired by ModuleFormer (Shen et al.,\n2023) that extends the concept of sparse activation to both the attention and feed-forward\nlayers. Unlike prior works that only apply sparse activation to the feed-forward layer,\nJetMoE-8B leverages sparse activation in both components to further reduce computational\ncosts while maintaining performance.\nImpressively, JetMoE-8B i", + "type": "text" + }, + { + "rank": 7, + "score": 0.0, + "content": "ive than generally\nthought. In addition, JetMoE-8B has 8B parameters while only activating 2B for each input\ntoken, reducing inference computation by about 70% compared to Llama2-7B.\nThe key advantages of JetMoE-8B include:\n•Openness and academia-friendly : JetMoE-8B is trained using only public datasets and\nopen-source training code, making it accessible to many academia research settings.\nThe mo", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "•Comprehensive open-source data mixture , which ensures high-quality training using\nonly open-source datasets.\nThese innovations in JetMoE-8B pave the way for more accessible and efficient LLMs, bene-\nfiting the broader AI research community. To foster collaboration and further advancements,\nwe have detailed all the training parameters and data mixture in this report.\n2 Model Architecture\n2.1 Mixt", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "SMoE, Shazeer et al. 2017). In this JetMoE, we use a linear layer to model the\nrouter\ns=Wrtrx, (1)\ng(e|x) =\u001a\nsoftmax (Topk(s))i,si∈Topk(s)\n0, si/∈Topk(s)(2)\nwhere Wrtris the expert embedding matrix of shape (N,Demb),Topkis the operator that\nselect the top klogits from s. The final output of the SMoE is then given by\ny=N\n∑\ne=1g(e|x)·fe(x) (3)\nWhen g(e|x) = 0,fe(x)will not need to be evaluated, thus", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "place FFD layers.\n2.2 FeedFoward Expert\nEach FFD expert is a standard 2-layer MLP with hidden state size Dffd:\nfmlp(x) =Woutσ(Winx) (4)\nWhere Woutis the output projection matrix of shape (Demb,Df f d),Winin the input projection\nmatrix of shape (2Df f d,Demb),σis the SwiGLU activation function.\n2.3 Attention Expert\nZhang et al. (2022) propose the Mixture of Attention heads (MoA), which extends SMOE", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "he number of attention head inside each attention experts,\nDheadis the dimension of each attention head. Among these matrices, We\nqand We\noare\nowned by each expert, but Wkand Wvare shared across experts to improve the training\nand inference efficiency.\nGiven an input vector sequence x, we first projected it to key vectors kand value vectors v\nusing the shared key and value projection matrices:\nk=W", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "r with more attention experts\nwhile maintaining the same amount of computation. Such that the attention layer will not\nbecome a performance bottleneck, while we scale up the MLP layers.\n3\n\nJetMoE\n2.4 Load Balancing during Pretraining\nTo avoid the SMoE repeatedly using the same module and wasting the extra capacity in\nthe other modules, it requires various load balancing losses to regulate the trai", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "the router probability allocated for expert i. To improve the training stability,\nwe also use the router z-loss introduced in Zoph et al. (2022):\nloss z=1\nBB\n∑\ni=1 \nlogN\n∑\nj=1exp(xi\nj)!2\n(11)\nwhere Bis the number of tokens, xis the logits given by router. The final training loss will\nbe the weighted sum of three losses:\nloss=loss lm+αloss b+βloss z (12)\nwhere αis the weight for load balancing loss", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "token extract of RefinedWeb publicly\navailable.\nStarCoder training data is sourced from The Stack v1.2 with code from GitHub spanning\n86 programming languages (Li et al., 2023b). The data is preprocessed through visual\ninspection, filtering, deduplication, and reweighting low-data languages. A new version of\nthe dataset has been recently released (Lozhkov et al., 2024).\nDolma is a large, open, div", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "encyclopedic content from\nWikipedia and Wikibooks (Soldaini et al., 2024).\nThe Pile is an 825 GB open-source English text corpus for training large language mod-\nels (Gao et al., 2020). It includes 22 diverse, publicly available datasets such as Wikipedia,\nNIH exPORTER, ArXiv, Books3, BookCorpus2, OpenSubtitles, YTSubtitles, and Enron\nEmails.\n3.1.1 Miscellaneous\n◦Proof-Pile-2 is a 55 billion token", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "atical web text (Paster et al., 2023).\n1http://commoncrawl.org/\n4\n\nJetMoE\n◦StackMathQA is a meticulously curated collection of 2 million mathematical questions\nand answers, sourced from various Stack Exchange sites (Zhang, 2024).\n◦OpenAssistant is a human-generated, human-annotated assistant-style conversation\ncorpus in 35 different languages. The corpus is a product of a worldwide crowd-\nsourcing", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "ity\ncommit messages on public Github repos that resemble natural language instruc-\ntions (Muennighoff et al., 2023a).\n3.2 Synthetic Datasets\nOpenHermes 2.5 is a large-scale, diverse, high-quality compilation of open-source and\ncustom synthetic datasets (Teknium, 2023). It contains 1 million primarily synthetically\ngenerated instruction and chat samples, following a ShareGPT structure. The dataset ", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "., 2023a), Glaive Code Assistant (glaiveai, 2023), GPT4-\nLLM (Peng et al., 2023), GPTeacher (Teknium1, 2023), Medical Tasks (CogStack, 2023),\nMetaMath 40k (Yu et al., 2023), SlimOrca 550K (Longpre et al., 2023; Mukherjee et al., 2023;\nLian et al., 2023), Platypus (Lee et al., 2024; Lightman et al., 2023; Wang et al., 2023b),\nShareGPT (GPT4-Only) (lm sys, 2023), and Unnatural Instructions GPT4 (Pen", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "trange-textbooks , and a select high-quality web\ncollection from math-ai/AutoMathText .\nUltraChat 200k is a filtered subset of the UltraChat dataset, which consists of 1.4M dialogues\ngenerated by ChatGPT (Ding et al., 2023; Tunstall et al., 2023b). The subset was created by\nselecting a smaller portion of the data, truecasing the text to fix grammatical errors, and\nremoving dialogues where the assi", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "r-OSS-75K datasets are generated using the\nOSS-INSTRUCT approach, which leverages a LLM to automatically create new coding\nproblems by drawing inspiration from random code snippets collected from open\nsource projects (Wei et al., 2023).\n◦Evol-Code Alpaca is an open-sourced implementation of Evol-Instruct adapted for\ncode instructions by streamlining, simplifying, and adding code-specific evolution", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/stone2022_chunks_ranked.json b/wattbot2025/data/ranked/stone2022_chunks_ranked.json new file mode 100644 index 00000000..7e014cf9 --- /dev/null +++ b/wattbot2025/data/ranked/stone2022_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.732792245357769, + "content": "ical procedures and hospital operations. But using this \ndata to enable more finely-grained diagnostics and treatments for both individual \npatients and patient populations has proved difficult. Research and deployment have \nbeen slowed by outdated regulations and incentive structures. Poor human-computer \ninteraction methods and the inherent difficulties and risks of implementing \ntechnologies i", + "type": "text" + }, + { + "rank": 2, + "score": 4.712975764631042, + "content": "0,1\n,\"promise in healthcare.61 The reduction or removal of these obstacles, combined with\"\n,\"innovations still on the horizon, have the potential to significantly improve health\"\n,outcomes and quality of life for millions of people in the coming years.\n,The clinical setting\n,\"For decades, the vision of an AI-powered clinician’s assistant has been a near cliché.\"\n,\"Although there have been succ", + "type": "table" + }, + { + "rank": 3, + "score": 4.635341962096172, + "content": "61 The reduction or removal of these obstacles, combined with \ninnovations still on the horizon, have the potential to significantly improve health \noutcomes and quality of life for millions of people in the coming years.\nThe clinical setting\nFor decades, the vision of an AI-powered clinician’s assistant has been a near cliché. \nAlthough there have been successful pilots of AI-related technol", + "type": "text" + }, + { + "rank": 4, + "score": 4.578774567359856, + "content": ",\"avoid spread of HIV. The dynamic, uncertain nature of these networks does pose\"\nuses of AI analytics is in,challenges for AI research.100 Care must also be taken to prevent AI systems from\n,\"reproducing discriminatory behavior, such as machine learning that identifies people\"\ndetecting white collar,\n,\"through illegal racial indicators, or through highly-correlated surrogate factors, such\"\n,\n\"c", + "type": "table" + }, + { + "rank": 5, + "score": 4.505464651359975, + "content": "ograms might be able to leverage homeless youth social networks to \nstrategically select peer leaders to spread health-related information, such as how to \navoid spread of HIV . The dynamic, uncertain nature of these networks does pose \nchallenges for AI research.100 Care must also be taken to prevent AI systems from \nreproducing discriminatory behavior, such as machine learning that identifies ", + "type": "text" + }, + { + "rank": 6, + "score": 4.33254671986449, + "content": "ess and manufacturing processes, or engaging with\"\n,\"privacy advocates or academics outside their walls, these companies viewed privacy as\"\n,\"a compliance activity. Their focus was on avoiding fines or punishments, rather than\"\n,proactively designing technology and adapting practices to protect privacy.\n,\"By contrast, the regulatory environment in the United States and Germany,\"\n,which combined mo", + "type": "table" + }, + { + "rank": 7, + "score": 4.296112911222336, + "content": "nishing interpersonal\"\n,interactions (entertainment). The report begins with a reflection on what constitutes\n,\"Artificial Intelligence, and concludes with recommendations concerning AI-related\"\n,policy. These recommendations include accruing technical expertise about AI in\n,government and devoting more resources—and removing impediments—to research\n,\"on the fairness, security, privacy, and societ", + "type": "table" + }, + { + "rank": 8, + "score": 4.15630594300979, + "content": "a reflection on what constitutes \nArtificial Intelligence, and concludes with recommendations concerning AI-related \npolicy . These recommendations include accruing technical expertise about AI in \ngovernment and devoting more resources—and removing impediments—to research \non the fairness, security , privacy , and societal impacts of AI systems.\nContrary to the more fantastic predictions for AI ", + "type": "text" + }, + { + "rank": 9, + "score": 4.1394672802921235, + "content": "liance activity . Their focus was on avoiding fines or punishments, rather than \nproactively designing technology and adapting practices to protect privacy .\nBy contrast, the regulatory environment in the United States and Germany , \nwhich combined more ambiguous goals with tough transparency requirements and \nmeaningful enforcement, were more successful in catalyzing companies to view \n138 Polit", + "type": "text" + }, + { + "rank": 10, + "score": 4.122764505778155, + "content": "d regulatory regimes will need to be\"\n,adapted to AI innovations or in some cases fundamentally reconfigured according to\nunderinvesting resources,\n,broadly accepted goals and principles.\n,\"The approach in the United States to date has been sector-specific, with oversight\"\nin research on the,\n,by a variety of agencies. The use of AI in devices that deliver medical diagnostics and\nsocietal implic", + "type": "table" + }, + { + "rank": 11, + "score": 4.106195981149432, + "content": "2003), 169-185.Ethical questions arise \nwhen programming cars to \nact in situations in which \nhuman injury or death is \ninevitable, especially when \nthere are split-second \nchoices to be made about \nwhom to put at risk.\n\n23AI is likely to have an increasing impact on city infrastructure. Accurate predictive \nmodels of individuals’ movements, their preferences, and their goals are likely to \nemerg", + "type": "text" + }, + { + "rank": 12, + "score": 4.0412325223029955, + "content": ", \nfairness, and other impacts of AI.\nQuestions include: Who is responsible when a self-driven car crashes or an \nintelligent medical device fails? How can AI applications be prevented from unlawful \ndiscrimination? Who should reap the gains of efficiencies enabled by AI technologies \nand what protections should be afforded to people whose skills are rendered obsolete? \nAs AI becomes integrated ", + "type": "text" + }, + { + "rank": 13, + "score": 4.0412325223029955, + "content": "0,1,2\nAI is likely to have an increasing impact on city infrastructure. Accurate predictive,,\n\"models of individuals’ movements, their preferences, and their goals are likely to\",,\nemerge with the greater availability of data. The ethical issues regarding such an,,\nemergence are discussed in Section III of this report.,,\nThe United States Department of Transportation released a call for propos", + "type": "table" + }, + { + "rank": 14, + "score": 0.0, + "content": "ARTIFICIAL INTELLIGENCE\nAND LIFE IN 2030\nONE HUNDRED YEAR STUDY ON ARTIFICIAL INTELLIGENCE | REPORT OF THE 2015 STUDY PANEL | SEPTEMBER 2016\nPREFACE\nThe One Hundred Year Study on \nArtificial Intelligence, launched \nin the fall of 2014, is a long-\nterm investigation of the field of \nArtificial Intelligence (AI) and \nits influences on people, their \ncommunities, and society . It \nconsiders the sc", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "e immediately prior report, \nenvisions the potential advances that lie ahead, and describes the technical and \nsocietal challenges and opportunities these advances raise, including in such arenas as \nethics, economics, and the design of systems compatible with human cognition. The \noverarching purpose of the One Hundred Year Study’s periodic expert review is to \nprovide a collected and connected", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "systems broadly benefit individuals and society .1 \nThe One Hundred Year Study is modeled on an earlier effort informally known as \nthe “ AAAI Asilomar Study .” During 2008-2009, the then president of the Association \nfor the Advancement of Artificial Intelligence (AAAI), Eric Horvitz, assembled a \ngroup of AI experts from multiple institutions and areas of the field, along with \nscholars of ", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "report on the intensive \nmeeting discussions, amplified by the participants’ subsequent discussions with other \ncolleagues, generated widespread interest and debate in the field and beyond. \nThe impact of the Asilomar meeting, and important advances in AI that included \nAI algorithms and technologies starting to enter daily life around the globe, spurred \nthe idea of a long-term recurring study ", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "One Hundred Year \nStudy’s periodic expert \nreview is to provide a \ncollected and connected \nset of reflections about \nAI and its influences as \nthe field advances.\n\n2extended deep thought and cross-disciplinary scholarly investigations that could \ninspire innovation and provide intelligent advice to government agencies and industry .\nThis report is the first in the planned series of studies tha", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "xperts in AI from academia, corporate laboratories \nand industry , and AI-savvy scholars in law, political science, policy , and economics, \nwas launched in mid-fall 2015. The participants represent diverse specialties and \ngeographic regions, genders, and career stages. \nThe Standing Committee extensively discussed ways to frame the Study Panel \ncharge to consider both recent advances in AI and p", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "ing or natural language processing, and studying particular \napplication areas such as healthcare or transportation. The committee ultimately \nchose a thematic focus on “ AI and Life in 2030” to recognize that AI’s various uses \nand impacts will not occur independently of one another, or of a multitude of other \nsocietal and technological developments. Acknowledging the central role cities have", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/strubell2019_chunks_ranked.json b/wattbot2025/data/ranked/strubell2019_chunks_ranked.json new file mode 100644 index 00000000..a07abef7 --- /dev/null +++ b/wattbot2025/data/ranked/strubell2019_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.721322204801323, + "content": "to\ndeter escalating rates of natural disaster, and based\non the estimated CO 2emissions listed in Table 1,\n1Sources: (1) Air travel and per-capita consump-\ntion:https://bit.ly/2Hw0xWc ; (2) car lifetime:\nhttps://bit.ly/2Qbr0w1 .\n\nmodel training and development likely make up\na substantial portion of the greenhouse gas emis-\nsions attributed to many NLP researchers.\nTo heighten the awareness of the", + "type": "text" + }, + { + "rank": 2, + "score": 4.1024634006709455, + "content": "0,1\nnity to this issue and promote mindful practice and,\"Amazon-AWS\n17%\n24%\n30%\n26%\"\n\"policy, we characterize the dollar cost and carbon\",\"Google\n56%\n14%\n15%\n10%\"\nemissions that result from training the neural net-,\"Microsoft\n32%\n23%\n31%\n10%\"\n\"works\nat\nthe core of many state-of-the-art NLP\",\nmodels. We do this by estimating the kilowatts,Table 2: Percent energy sourced from: Renewable (e.g.\nof ene", + "type": "table" + }, + { + "rank": 3, + "score": 3.7214060540838116, + "content": "urces are available, model training also incurs a\nsubstantial cost to the environment due to the en-\nergy required to power this hardware for weeks or\nmonths at a time. Though some of this energy may\ncome from renewable or carbon credit-offset re-\nsources, the high energy demands of these models\nare still a concern since (1) energy is not currently\nderived from carbon-neural sources in many loca-\n", + "type": "text" + }, + { + "rank": 4, + "score": 3.3195804652191327, + "content": "equired to train a variety of popular,\"hydro, solar, wind), natural gas, coal and nuclear\nfor\"\n,\"the top 3 cloud compute providers (Cook et al., 2017),\"\n\"off-the-shelf NLP models, which can be converted\",\n,\"compared to the United States,4 China5 and Germany\"\n\"to approximate\ncarbon emissions\nand electricity\",\n,\"(Burger, 2019).\"\n\"costs.\nTo estimate the even greater\nresources re-\",\n\"quired to transfe", + "type": "table" + }, + { + "rank": 5, + "score": 3.2816227769237316, + "content": "of popular\noff-the-shelf NLP models, which can be converted\nto approximate carbon emissions and electricity\ncosts. To estimate the even greater resources re-\nquired to transfer an existing model to a new task\nor develop new models, we perform a case study\nof the full computational resources required for the\ndevelopment and tuning of a recent state-of-the-art\nNLP pipeline ( Strubell et al. ,2018 ).", + "type": "text" + }, + { + "rank": 6, + "score": 3.1727852929933134, + "content": "0,1\n\"4.1\nCost of training\",\n,\"1\n120\n$52–$175\n$5\"\nTable 3 lists CO2 emissions and estimated cost of,\n,\"24\n2880\n$1238–$4205\n$118\"\ntraining the models described in §2.1. Of note is,\n,\"4789\n239,942\n$103k–$350k\n$9870\"\n\"that TPUs are more cost-efficient\nthan GPUs on\",\n,Table 4: Estimated cost in terms of cloud compute and\nworkloads that make sense for that hardware (e.g.,\n,\"electricity for training:\n(1) ", + "type": "table" + }, + { + "rank": 7, + "score": 2.7309185955564477, + "content": "t\ntraining a neu-\"\n\"a result,\ntraining a state-of-the-art model now re-\",ral network might better be allocated to heating a\n\"quires substantial computational\nresources which\",\"family’s home.\nIt\nis estimated that we must cut\"\n\"demand considerable\nenergy,\nalong with the as-\",carbon emissions by half over the next decade to\n\"sociated financial and environmental\ncosts.\nRe-\",\"deter escalating rates of n", + "type": "table" + }, + { + "rank": 8, + "score": 2.38766310725761, + "content": "7Power\nand carbon footprint are omitted for TPUs due to lack of publi c information on power draw for this hardware.\n4 Experimental results\n4.1 Cost of training\nTable 3lists CO 2emissions and estimated cost of\ntraining the models described in §2.1. Of note is\nthat TPUs are more cost-efficient than GPUs on\nworkloads that make sense for that hardware (e.g.\nBERT). We also see that models emit substan-", + "type": "text" + }, + { + "rank": 9, + "score": 1.8393231457201773, + "content": "ump-\",\n,\"Transformer (big)\n192\"\n\"tion.\nAs a result\nthese models are costly to\",\n,\"w/ neural architecture search\n626,155\"\n\"train and develop, both financially, due to the\",\ncost of hardware and electricity or cloud com-,\n,Table 1: Estimated CO2 emissions from training com-\n\"pute time, and environmentally, due to the car-\",\n,\"mon NLP models, compared to familiar consumption.1\"\n\"bon footprint\nrequired", + "type": "table" + }, + { + "rank": 10, + "score": 1.8253020055246485, + "content": "0,1\n\"three\ncloud service providers.\nThe U.S. break-\",\"(2019)\nreport\nthat\nthe BERT\nence. Devlin et al.\"\ndown of energy is comparable to that of the most,base model (110M parameters) was trained on 16\n\"popular cloud compute service, Amazon Web Ser-\",TPU chips for 4 days (96 hours). NVIDIA reports\n\"vices, so we believe this conversion to provide a\",that they can train a BERT model in 3.3 days (79.2\nr", + "type": "table" + }, + { + "rank": 11, + "score": 1.8114930141175312, + "content": "0,1\n\"model\ntraining and development\nlikely make up\",\"Consumer\nRenew.\nGas\nCoal\nNuc.\"\na substantial portion of the greenhouse gas emis-,\"China\n22%\n3%\n65%\n4%\"\nsions attributed to many NLP researchers.,\"Germany\n40%\n7%\n38%\n13%\"\nTo heighten the awareness of the NLP commu-,\"United States\n17%\n35%\n27%\n19%\"\nnity to this issue and promote mindful practice and,\"Amazon-AWS\n17%\n24%\n30%\n26%\"\n\"policy, we characte", + "type": "table" + }, + { + "rank": 12, + "score": 1.4904047895627837, + "content": "ty of re-,\n,multiple instances of specialized hardware such as\n\"cently successful neural network models\nfor\",\n\"NLP. Based on these findings, we propose ac-\",\"GPUs or TPUs,\ntherefore limiting access to these\"\ntionable recommendations to reduce costs and,highly accurate models on the basis of finances.\nimprove equity in NLP research and practice.,\n,\"Even when these expensive computational\nre-\"\n,\"sourc", + "type": "table" + }, + { + "rank": 13, + "score": 1.4675075654987138, + "content": "s energy may\n\"abled impressive\naccuracy improvements\nacross\",\"come from renewable or carbon credit-offset\nre-\"\n\"al.,\nmany fundamental NLP tasks\n(Bahdanau et\",\"sources, the high energy demands of these models\"\n\"2015;\nLuong\net\nal.,\n2015; Dozat\nand Man-\",are still a concern since (1) energy is not currently\n\"2017; Vaswani\net\nal.,\n2017),\nwith\nthe\nning,\",derived from carbon-neural sources in many loca-", + "type": "table" + }, + { + "rank": 14, + "score": 1.4507910891074935, + "content": "self-attention models that\nhave become commonplace in NLP, nor do they\nextrapolate power to estimates of carbon and dol-\nlar cost of training.\nAnalysis of hyperparameter tuning has been\nperformed in the context of improved algorithms\nfor hyperparameter search ( Bergstra et al. ,2011 ;\nBergstra and Bengio ,2012 ;Snoek et al. ,2012 ). To\nour knowledge there exists to date no analysis of\nthe computat", + "type": "text" + }, + { + "rank": 15, + "score": 1.4290860040061673, + "content": "ptible ($1.46/hr–$2.40/ hr)\nand on-demand ($4.50/hr–$8/hr) pricing as lower and upper\nbounds for TPU v2/3; cheaper bulk contracts are available.\n\nModel Hardware Power (W) Hours kWh ·PUE CO 2e Cloud compute cost\nTransformer base P100x8 1415.78 12 27 26 $41–$140\nTransformer big P100x8 1515.43 84 201 192 $289–$981\nELMo P100x3 517.66 336 275 262 $433–$1472\nBERTbase V100x64 12,041.51 79 1507 1438 $3751", + "type": "text" + }, + { + "rank": 16, + "score": 1.4290860040061673, + "content": "pute time and non-trivial carbon emissions.\n4.2 Cost of development: Case study\nTo quantify the computational requirements of\nR&D for a new model we study the logs of\nall training required to develop Linguistically-\nInformed Self-Attention ( Strubell et al. ,2018 ), a\nmulti-task model that performs part-of-speech tag-\nging, labeled dependency parsing, predicate detec-\ntion and semantic role labeli", + "type": "text" + }, + { + "rank": 17, + "score": 1.4290860040061673, + "content": "e and gigaflops\n\"base model\nrequires 10 hours\nto train for 300k\",\n,required during inference. They also measure av-\n\"steps on one TPUv2 core. This equates to 32,623\",\n,erage power draw required during inference on\n\"hours of TPU or 274,120 hours on 8 P100 GPUs.\",\n,GPUs as a function of batch size. Neither work an-\n\"ELMo.\nThe ELMo model\n(Peters et al., 2018)\",\n,alyzes the recurrent and self-attention", + "type": "table" + }, + { + "rank": 18, + "score": 1.397719374247828, + "content": "constantly throughout\nthe\"\nof-the-art BLEU score of 29.7 for English to Ger-,\n,6 month duration of the project. Table 4 lists upper\n\"man machine translation,\nan increase of\njust 0.1\",\n,\"and lower bounds of\nthe estimated cost\nin terms\"\n\"BLEU at\nthe cost of at\nleast $150k in on-demand\",\n,of Google Cloud compute and raw electricity re-\ncompute time and non-trivial carbon emissions.,\n,quired to develo", + "type": "table" + }, + { + "rank": 19, + "score": 0.0, + "content": "arXiv:1906.02243v1 [cs.CL] 5 Jun 2019Energy and Policy Considerations for Deep Learning in NLP\nEmma Strubell Ananya Ganesh Andrew McCallum\nCollege of Information and Computer Sciences\nUniversity of Massachusetts Amherst\n{strubell, aganesh, mccallum }@cs.umass.edu\nAbstract\nRecent progress in hardware and methodol-\nogy for training neural networks has ushered\nin a new generation of large networks ", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "models are costly to\ntrain and develop, both financially, due to the\ncost of hardware and electricity or cloud com-\npute time, and environmentally, due to the car-\nbon footprint required to fuel modern tensor\nprocessing hardware. In this paper we bring\nthis issue to the attention of NLP researchers\nby quantifying the approximate financial and\nenvironmental costs of training a variety of re-\ncently s", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/wu2021a_chunks_ranked.json b/wattbot2025/data/ranked/wu2021a_chunks_ranked.json new file mode 100644 index 00000000..510f97a0 --- /dev/null +++ b/wattbot2025/data/ranked/wu2021a_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 6.144959070781123, + "content": "el quantization for RMs\n[34].\",\n,centers with 100% renewable energy purchased by Facebook.\n\"Quantization offers two primary efficiency benefits:\nthe low-\",\n,\"Remaining emissions\nare offset with various\nsustainability\"\nprecision data representation reduces the amount of compu-,\n,\"programs,\nfurther\nreducing the operational carbon footprint of\"\n\"tation requirement and, at\nthe same time,\nlowers the over", + "type": "table" + }, + { + "rank": 2, + "score": 4.992962269311858, + "content": "sis (LCA) is a common methodology to\nassess the carbon emissions over the product life cycle. There\nare four major phases: manufacturing ,transport ,product use ,\nandrecycling2. From the perspective of AI’s carbon footprint\nanalysis, manufacturing andproduct use are the focus. Thus,\nin this work, we consider the overall carbon footprint of\nAI by including manufacturing — carbon emissions from\nbuil", + "type": "text" + }, + { + "rank": 3, + "score": 4.087634410442533, + "content": "ower\nreduction across Facebook’s AI\",\"tion opportunities available with judicious cross-stack, hard-\"\n\"fleet over 6-month period from each of\nthe optimization areas.\",\"ware/software\noptimization.\nIn\naddition\nto\noptimizing\nthe\"\n\"The optimizations\nin aggregate provide, on average, a 20%\",\"carbon footprint\nfor\nthe language translation task, we describe\"\nreduction in operational power consumption every", + "type": "table" + }, + { + "rank": 4, + "score": 4.0836215452131475, + "content": "arbon footprint\n\"Processing phase\nand apply weights\nto individual\nfeatures\",\n,\"analysis, manufacturing and product use are the focus. Thus,\"\nbased on feature importance to the model optimization objective.,\n,\"in\nthis work, we\nconsider\nthe\noverall\ncarbon\nfootprint\nof\"\n\"During Experimentation,\nthe researchers design,\nimplement\",\n,\"AI by including manufacturing — carbon emissions\nfrom\"\n\"and evaluate ", + "type": "table" + }, + { + "rank": 5, + "score": 3.986616277761611, + "content": "cance of embodied carbon emissions using\nFacebook’s Greenhouse Gas (GHG) emission statistics3.In this\ncase, more than 50% of Facebook’s emissions owe to its value\nchain — Scope 3 of Facebook’s GHG emission . As a result,\na significant embodied carbon cost is paid upfront for every\nsystem component brought into Facebook’s fleet of datacenters,\nwhere AI is the biggest growth driver.\n2Recycling is an i", + "type": "text" + }, + { + "rank": 6, + "score": 3.915340630274342, + "content": "across\nfrom [21] and may not be reflective of production use cases.,\n,ML use cases.\n,\"The overall operational carbon footprint\nis categorized into\"\n,\"offline training, online training, and inference. Offline training\"\nIII. AI COMPUTING’S CARBON FOOTPRINT,\n,encompasses both experimentation and training models with\n,\"historical\ndata. Online\ntraining\nis\nparticularly\nrelevant\nto\"\nA. Carbon Footprint Anal", + "type": "table" + }, + { + "rank": 7, + "score": 3.8815988792819254, + "content": "he use of AI\n(i.e., operational carbon footprint).\"\nis computationally-intensive. A large collection of diverse ML,\n,While quantifying the exact breakdown between operational\n\"ideas are explored simultaneously at-scale. Thus, during this\",\n,\"and\nembodied\ncarbon\nfootprint\nis\na\ncomplex\nprocess, we\"\n\"phase, we observe unique system resource requirements from\",\n,estimate the significance of embodied ca", + "type": "table" + }, + { + "rank": 8, + "score": 3.856173902598764, + "content": "usly\nupdated based on recent data. The inference footprint represents\nthe emission from serving production traffic. The online training\nand inference emissions are considered over the period of\noffline training. For recommendation use cases, we find the\ncarbon footprint is split evenly between training and inference.\nOn the other hand, the carbon footprint of LM is dominated\nby the inference phase, u", + "type": "text" + }, + { + "rank": 9, + "score": 3.7910779035385667, + "content": "server utilization exhibits\na diurnal pattern, Auto-Scaling frees the over-provisioned\ncapacity during off-peak hours, by up to 25% of the web\ntier’s machines [ 38]. By doing so, it provides opportunistic\nserver capacity for others to use, including offline ML training.\nFurthermore, static power consumption plays a non-trivial role\nin the context of the overall data center electricity footprint.\nTh", + "type": "text" + }, + { + "rank": 10, + "score": 3.77751727947952, + "content": "arbon cost\nas the dominating source of AI’s carbon footprint.\nB. Carbon Footprint Optimization from Hardware-Software\nCo-Design\nOptimization is an iterative process — we reduce the power\nfootprint across the machine learning hardware-software stack\nby 20% every 6 months. But at the same time, AI infrastructure\ncontinued to scale out. The net effect, with Jevon’s Paradox, is\na 28.5% operational pow", + "type": "text" + }, + { + "rank": 11, + "score": 3.6554762835070775, + "content": "urce-efficient models), platform\n(e.g., PyTorch’s support for quantization), infrastructure (e.g.,\ndata center optimization and low-precision hardware), and\nhardware (e.g., domain-specific acceleration). Each bar illus-\ntrates the operational power reduction across Facebook’s AI\nfleet over 6-month period from each of the optimization areas.\nThe optimizations in aggregate provide, on average, a 20%\nre", + "type": "text" + }, + { + "rank": 12, + "score": 3.566749583896587, + "content": "\"footprint\nreduction across Facebook’s AI fleet over\ntwo years.\",improvement on GPUs. Another 5× energy efficiency gain\n\"The\nimprovement\ncome\nfrom four\nareas\nof\noptimizations:\",can be achieved by using custom operators to schedule\n\"model\n(e.g., designing resource-efficient models), platform\",\"encoding steps within a single kernel of\nthe Transformer\"\n\"(e.g., PyTorch’s support\nfor quantization),\ninfras", + "type": "table" + }, + { + "rank": 13, + "score": 3.4515560799197975, + "content": "U acceleration. In addition to caching, deploying LM\"\n\"by\n20% every 6 months. But at the same time, AI infrastructure\",\n,across GPU-based specialized AI hardware unlocks an\n\"continued to scale out. The net effect, with Jevon’s Paradox,\nis\",\n,additional 10.1× energy efficiency improvement.\na 28.5% operational power footprint reduction over two years,\n,\"• Algorithmic optimization. Finally, algorithmi", + "type": "table" + }, + { + "rank": 14, + "score": 3.4114909186335907, + "content": "recommendation\nand ranking use cases, the embedding operation dominates the\ninference execution time [27], [33].\nTo tackle the significant memory capacity and bandwidth\nrequirement, we deploy model quantization for RMs [ 34].\nQuantization offers two primary efficiency benefits: the low-\nprecision data representation reduces the amount of compu-\ntation requirement and, at the same time, lowers the ove", + "type": "text" + }, + { + "rank": 15, + "score": 3.2018066382644874, + "content": "sustainable AI. From the\nsystem’s perspective, the life cycle of model development and\n7Papers with code: https://paperswithcode.com/sota/image-classification-on\n-imagenet\n8https://github.com/mlcommons/algorithmic-efficiency/\n9https://2021.naacl.org/ethics/faq/#-if-my-paper-reports-on-experiments-t\nhat-involve-lots-of-compute-timepowersystem hardware, including manufacturing andoperational use ,\nmus", + "type": "text" + }, + { + "rank": 16, + "score": 3.021071338017019, + "content": "rimentation and training toinference .\nWe characterize the carbon footprint of AI computing by\nexamining the model development cycle across industry-scale\nmachine learning use cases at Facebook (Section II). This is\nillustrated by the more than 800 \u0002operational carbon footprint\nreduction achieved through judicious hardware-software co-\ndesign for a Transformer-based universal language model.\nTakin", + "type": "text" + }, + { + "rank": 17, + "score": 3.009898482715895, + "content": "yond its operational\nexamining the model development cycle across industry-scale\"\n\"energy consumption. The embodied carbon footprint of systems\nmachine learning use cases at Facebook (Section II). This is\"\n\"is becoming a dominating factor for AI’s overall environmental\nillustrated by the more than 800× operational carbon footprint\"\n\"impact\n(Section III)\n[19].\nreduction achieved through judicious h", + "type": "table" + }, + { + "rank": 18, + "score": 2.9916590427037884, + "content": "ting design opportunities (Section IV-C).\nThe growth of AI in all dimensions outpaces the efficiency im-\nprovement at-scale. Figure 9 illustrates that, as GPU utilization\nis improved (x-axis) for LM training on GPUs, both embodied\nand operational carbon emissions will reduce. Increasing GPU\nutilization up to 80%, the overall carbon footprint decreases\nby 3\u0002. Powering AI services with renewable ener", + "type": "text" + }, + { + "rank": 19, + "score": 2.9770425346166447, + "content": "on. In addition to optimizing the\ncarbon footprint for the language translation task, we describe\nadditional optimization techniques tailored for ranking and\n5\n\nPerformance-per-Watt[Domain-Specific Acceleration]Utilization[At-Scale Data Center Optimization;Low-Precision Hardware]0.80.911.11.21.3\nYr1-H1Yr1-H2Yr2-H1Yr2-H2Operational Power FootprintBaselineOptimized (Section 3.2)28.5% improvement Fig", + "type": "text" + }, + { + "rank": 20, + "score": 2.965676074190048, + "content": "ation\nmodels, require significantly higher memory capacity and\nbandwidth [ 55], [33]. This motivates researchers to develop\nmemory-efficient model architectures. For example, the Tensor-\nTrain compression technique (TT-Rec) achieves more than\n100\u0002memory capacity reduction with negligible training time\nand accuracy trade-off [ 56]. Similarly, the design space trade-\noff between memory capacity requir", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/wu2021b_chunks_ranked.json b/wattbot2025/data/ranked/wu2021b_chunks_ranked.json new file mode 100644 index 00000000..ba9a9f74 --- /dev/null +++ b/wattbot2025/data/ranked/wu2021b_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 9.408054669931193, + "content": "actual generation to better utilize the energy [Elkin and\nWitherspoon]. Another example is the current Covid outbreak. As horrendous as the outbreak has been and continues\nto be, technology has enabled a portion of the economy to continue to operate even as employees work remotely. In\naddition, the global carbon emissions for 2020 dropped by 6.4% with vehicle transportation in the US accounting fo", + "type": "text" + }, + { + "rank": 2, + "score": 8.432527667134746, + "content": "n emissions for 2020 dropped by 6.4% with vehicle transportation in the US accounting for a\",\n\"portion of the global reduction [Tollefson, 2021]. Looking forward, information technology can improve efficiencies\",\n\"in practically every sector, from manufacturing to food production to transportation to controlling the climate in our\",\n\"homes and offices. Although there is a carbon cost associated with", + "type": "table" + }, + { + "rank": 3, + "score": 7.583906387050217, + "content": "r products. Examples of such commitments are Facebook achieving NetZero in operational emissions\nin 2020 and across its value chain by 2030 [Facebook], Apple’s pledge for 100% carbon neutral supply chain by\n2030 [Apple], Microsoft’s goal of being carbon negative by 2030 [Smith], and Google’s aim of 24x7 carbon free data\ncenters [Google, b]. In 2020, Amazon, Google, Facebook, and Microsoft were the", + "type": "text" + }, + { + "rank": 4, + "score": 6.170585897195081, + "content": "ust a few of the topics under the larger sustainability umbrella. There is increased focus on\nthese topics from both industrial and political institutions. All major technology companies have pledged to reduce or\neliminate their carbon footprint in the next decade by reducing the environmental impact associated with manufacturing\nand using their products. Examples of such commitments are Facebook ", + "type": "table" + }, + { + "rank": 5, + "score": 6.132741879618505, + "content": "the\"\n\"International Symposium on Computer Architecture, 2021.\"\n\"Facebook. Facebook is committed to reaching net zero emissions across our value chain in 2030, aligning our efforts\"\nwith the latest science on what is needed to transition to a zero-carbon future.\nhttps://sustainability.fb.com/report-page/climate/.\nApple. Apple commits to be 100 percent carbon neutral for its supply chain and product", + "type": "table" + }, + { + "rank": 6, + "score": 5.839987650250076, + "content": "0, Amazon, Google, Facebook, and Microsoft were the top four technology companies\"\n\"that purchased significant renewable energy capacities, accounting for 30% of the cumulative total from corporations\"\n\"globally [Schechner, 2021].\nIn addition, countries and trading zones are legislating carbon emission requirements.\"\n\"China has committed to be carbon free by 2060 [Myers, 2020] and the EU has commit", + "type": "table" + }, + { + "rank": 7, + "score": 5.3087370262693145, + "content": "ple.com/newsroom/2020/07/\napple-commits-to-be-100-percent-carbon-neutral-for-its-supply-chain-and-products-by-2030/ .\nBrad Smith. Microsoft will be carbon negative by 2030.\nhttps://blogs.microsoft.com/blog/2020/01/16/microsoft-will-be-carbon-negative-by-2030/ .\nGoogle. The Internet is 24x7—carbon-free energy should be too.\nhttps://sustainability.google/progress/projects/24x7/ , b.\nSam Schechner. A", + "type": "text" + }, + { + "rank": 8, + "score": 5.242733573983652, + "content": "by 2060 [Myers, 2020] and the EU has committed to cut carbon emissions by\n55% by 2030 [BBC, 2021].\nSustainability targets and the associated regulations will continue to increase, and hardware must be manufactured with\nless planetary impact, use less energy while in operation, and produce less e-waste at the end of life [Orcuttarchive, 2015,\nChang et al., 2017]. Existing practices such as the move", + "type": "text" + }, + { + "rank": 9, + "score": 4.8635777411308805, + "content": "t Gupta. Most of computing’s carbon emissions are coming from manufacturing and\n\"infrastructure. https://tech.fb.com/sustainable-computing/, 2021.\"\n\"Udit Gupta, Young Geun Kim, S. Lee, J. Tse, Hsien-Hsin S. Lee, Gu-Yeon Wei, D. Brooks, and Carole-Jean Wu.\"\nChasing carbon: The elusive environmental footprint of computing. Proceedings of the IEEE International\n\"Symposium on High-Performance Computer", + "type": "table" + }, + { + "rank": 10, + "score": 4.731330608341983, + "content": "sakis, Deli Zhang, Pulkit Misra, Rod Assis, Kyle Woolcock, Nithish\nMahalingam, Brijesh Warrier, David Gauthier, Lalu Kunnath, Steve Solomon, Osvaldo Morales, Marcus Fontoura,\nand Ricardo Bianchini. Flex: High-availability datacenters with zero reserved power. In Proceedings of the\nInternational Symposium on Computer Architecture , 2021.\nFacebook. Facebook is committed to reaching net zero emission", + "type": "text" + }, + { + "rank": 11, + "score": 4.731330608341983, + "content": "and Transparency, page 610–623, 2021.\"\nJon Kleinberg and Manish Raghavan. Algorithmic monoculture and social welfare. Proceedings of the National\n\"Academy of Sciences, 118(22), 2021.\"\n\"Ravi Jain and John Wullert. Challenges: Environmental design for pervasive computing systems.\nIn Proceedings of the\"\n\"8th Annual International Conference on Mobile Computing and Networking, page 263–270, 2002.\"\n\"Emm", + "type": "table" + }, + { + "rank": 12, + "score": 4.731330608341983, + "content": "ergy storage, 2020.\"\nCarl Elkin and Sims Witherspoon. Machine learning can boost the value of wind energy.\nhttps://deepmind.com/blog/article/machine-learning-can-boost-value-wind-energy.\nJeff Tollefson. COVID curbed carbon emissions in 2020 — but not by much.\n\"https://www.nature.com/articles/d41586-021-00090-3, 2021.\"\n\"Jichuan Chang, Justin Meza, P. Ranganathan, C. Bash, and Amip Shah. Green serve", + "type": "table" + }, + { + "rank": 13, + "score": 4.68883208616153, + "content": "obile Computing and Networking , page 263–270, 2002.\nEmma Strubell, Ananya Ganesh, and Andrew McCallum. Energy and policy considerations for deep learning in nlp,\n2019.\nSrilatha Manne. Examining the Carbon Footprint of Devices. https://devblogs.microsoft.com/\nsustainable-software/examining-the-carbon-footprint-of-devices/ , 2020.\nCarole-Jean Wu and Udit Gupta. Most of computing’s carbon emissions ", + "type": "text" + }, + { + "rank": 14, + "score": 4.68883208616153, + "content": "icity/use-of-electricity.php .\nC. Lawrence Zitnick, Lowik Chanussot, Abhishek Das, Siddharth Goyal, Javier Heras-Domingo, Caleb Ho, Weihua Hu,\nThibaut Lavril, Aini Palizhati, Morgane Riviere, Muhammed Shuaibi, Anuroop Sriram, Kevin Tran, Brandon Wood,\nJunwoong Yoon, Devi Parikh, and Zachary Ulissi. An introduction to electrocatalyst design using machine learning\nfor renewable energy storage, 2020.", + "type": "text" + }, + { + "rank": 15, + "score": 4.526207941277976, + "content": "silient computing\ninfrastructures can often come with significant environmental implications. Hence, any computing infrastructure\nsolutions must be cognizant of the multifaceted nature of the problems being addressed.\nSustainability. Resource limitations, climate change, water depletion, electronic waste, ecosystem damage, and\nenvironmental racism are just a few of the topics under the larger susta", + "type": "text" + }, + { + "rank": 16, + "score": 4.232606466698837, + "content": "0,1\nSocio-Technological Challenges and Opportunities: Paths Forward,A PREPRINT\n\"Adminstration]. This is where smart home IoT devices, such as Nest, can have an impact. AI is used to discover new\",\n\"electrocatalysts for more efficient and scalable ways to store and use renewable energy [Zitnick et al., 2020] while also\",\nbeing used to predict renewable energy availability ahead of actual generation ", + "type": "table" + }, + { + "rank": 17, + "score": 4.1824160032896485, + "content": "intermittent nature of renewable energy generation but the cost must be\n\"significantly improved for practical deployment [Plumer, 2021]. And, for technology to be truly inclusive, the way AI\"\n\"technologies are developed and used must be human-centered, driven by the cultural and demographic differences in\"\n\"the population and with pro-social goals [Stray, 2021]. Furthermore, given its increasingly ", + "type": "table" + }, + { + "rank": 18, + "score": 4.114864687142194, + "content": "ion but the cost must be\nsignificantly improved for practical deployment [Plumer, 2021]. And, for technology to be truly inclusive, the way AI\ntechnologies are developed and used must be human-centered, driven by the cultural and demographic differences in\nthe population and with pro-social goals [Stray, 2021]. Furthermore, given its increasingly large impact on the society,\nAI must be developed wi", + "type": "text" + }, + { + "rank": 19, + "score": 2.711844775419322, + "content": "ernet is 24x7—carbon-free energy should be too.\n\"https://sustainability.google/progress/projects/24x7/, b.\"\nSam Schechner. Amazon and Other Tech Giants Race to Buy Up Renewable Energy. https://www.wsj.com/\n\"articles/amazon-and-other-tech-giants-race-to-buy-up-renewable-energy-11624438894, 2021.\"\nSteven Lee Myers. China’s Pledge to Be Carbon Neutral by 2060: What It Means.\n\"https://www.nytimes.com/", + "type": "table" + }, + { + "rank": 20, + "score": 2.505873988842314, + "content": "a-climate-change.html , September 2020.\nBBC. Climate change: EU to cut CO2 emissions by 55% by 2030.\nhttps://www.bbc.com/news/world-europe-56828383 , April 2021.\nMike Orcuttarchive. A biodegradable computer chip that performs surprisingly well.\nhttps://www.technologyreview.com/2015/07/14/167161/\na-biodegradable-computer-chip-that-performs-surprisingly-well/ , 2015.\nTing-Jung Chang, Zhuozhi Yao, Pa", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/xia2024_chunks_ranked.json b/wattbot2025/data/ranked/xia2024_chunks_ranked.json new file mode 100644 index 00000000..f0274af3 --- /dev/null +++ b/wattbot2025/data/ranked/xia2024_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 4.022848561652206, + "content": "ba fine-tuning exhibited a slight reduction\nin latency as sequence length increased, with approximately\n19% and 25% decreases for sparse and dense fine-tuning,\nrespectively. This is due to the varying maximum batch sizes\nsupported by each sequence length, resulting in a similar\nnumber of tokens in each batch. Because latency remains\nconsistent with increasing sequence length and we can use\nlarger ", + "type": "text" + }, + { + "rank": 2, + "score": 4.007100800600602, + "content": "gle GPU load balancing [30] and multi-GPU load\nbalancing [31] have been proposed to address this issue.Takeaway 6 .The effect of fine-tuning on expert\nload imbalance in the MoE layer is LLM model and\ndataset dependent.\n6) Sensitivity Study on Sequence Length: To further ana-\nlyze the effect of sequence length on the fine-tuning process,\nwe chose the batch size that would maximize the memory\nfor ea", + "type": "text" + }, + { + "rank": 3, + "score": 0.0, + "content": "Understanding the Performance and Estimating the\nCost of LLM Fine-Tuning\nYuchen Xia1Jiho Kim2Yuhan Chen1Haojie Ye1Souvik Kundu3\nCong (Callie) Hao2Nishil Talati1\n1University of Michigan2Georgia Institute of Technology3Intel Labs\nAbstract —Due to the cost-prohibitive nature of training Large\nLanguage Models (LLMs), fine-tuning has emerged as an attrac-\ntive alternative for specializing LLMs for spec", + "type": "text" + }, + { + "rank": 4, + "score": 0.0, + "content": "se and dense versions of\nMoE models, as well as their runtime characteristics, including\nmaximum batch size, execution time breakdown, end-to-end\nthroughput, GPU hardware utilization, and load distribution.\nOur study identifies the optimization of the MoE layer as crucial\nfor further improving the performance of LLM fine-tuning.\nUsing our profiling results, we also develop and validate an\nanalytic", + "type": "text" + }, + { + "rank": 5, + "score": 0.0, + "content": "e Language Models (LLMs) are widely utilized in\nNatural Language Processing (NLP) [1]. Modern LLMs\ntypically possess billions to trillions of parameters, neces-\nsitating extensive time and resources for training. For in-\nstance, the estimated cost of training OpenAI’s GPT-4 model\nexceeds $100 million, rendering it financially prohibitive\nfor most small-to-medium size enterprises and the academic\nc", + "type": "text" + }, + { + "rank": 6, + "score": 0.0, + "content": "domain-specific dataset to align the desired behav-\niors of LLMs through supervised fine-tuning on instruction-\nfollowing tasks [6]. Unlike pre-training, fine-tuning can be\nconducted in a resource-constrained environment, typically\nusing one or a few GPUs. Consequently, fine-tuning presents\na compelling case for applications such as specialized ques-\ntion answering within enterprises, legal docume", + "type": "text" + }, + { + "rank": 7, + "score": 0.0, + "content": "st of fine-tuning on the cloud. Given\nour focus on cost-efficient LLM fine-tuning, we concen-\ntrate on fine-tuning sparse Mixture-of-Expert (MoE) models.\nSpecifically, we employ an attention-based MoE model, Mix-\ntral [4], and a state-space MoE model, BlackMamba [8]. Us-\ning these models and two domain-specific datasets for math-\nematics and common-sense question-answering, we conduct\nan in-depth ", + "type": "text" + }, + { + "rank": 8, + "score": 0.0, + "content": "single GPU memory budget, exe-\ncution time breakdown and bottlenecks, overall throughput,\nmicroarchitectural performance counters, and runtime load\ndistribution. The insights gained from our study are used to\ndevelop and validate an analytical model to estimate the cost.\nOur characterization uncovers the following unique in-\nsights. (1) Fine-tuning can be achieved in less than 10 epochs,\nand spars", + "type": "text" + }, + { + "rank": 9, + "score": 0.0, + "content": "-end throughput by supporting\na larger batch size. Given similar learning abilities of sparse\nand dense models, it is desired to use a sparse MoE model\nfor cost-effective fine-tuning. (4) The workload becomes\ncompute bound by increasing batch size; improving compute\nresources will increase performance. (5) Fine-tuning sparse\nmodel leads to more load imbalance.\nBased on these insights, we create an", + "type": "text" + }, + { + "rank": 10, + "score": 0.0, + "content": "n\n0.55. Using the estimated throughput, our model calculates\nthe fine-tuning cost for different cloud providers.\nThe contributions of this paper are as follows.\n•Make a case for LLM fine-tuning for specializing pre-\ntrained models in a cost-effective manner.arXiv:2408.04693v1 [cs.CL] 8 Aug 2024\n\nFig. 1. LLM model overview. We evaluate accuracy, throughput, runtime,\nand GPU characterization for d", + "type": "text" + }, + { + "rank": 11, + "score": 0.0, + "content": "nd validation of an analytical model to estimate\nthe cost of LLM fine-tuning in the cloud.\nII. B ACKGROUND\nA. LLM and Finetuning\nThe decoder-only Transformer is designed to handle tasks\nwhere the output generation depends solely on the preceding\ntokens, making it particularly suited for auto-regressive tasks\nsuch as language modeling and text generation [9]. In the\nclassic decoder-only Transformer", + "type": "text" + }, + { + "rank": 12, + "score": 0.0, + "content": "ded into several smaller\nFFNs, referred to as experts, which are sparsely activated\nby a gating mechanism. The self-attention block can also\nbe replaced with a Mamba layer to improve performance in\nsequence modeling (a model known as state-space model).\nLLMs like GPT [10], [11], LLaMA [3], Claude [12], Mis-\ntral [13] have demonstrated their ability to excel in many\nnatural language processing (NLP", + "type": "text" + }, + { + "rank": 13, + "score": 0.0, + "content": "ecific data, enabling it to understand\nand generate content that aligns closely with the unique\nneeds of the users. For instance, in the healthcare sector,\na fine-tuned LLM can assist in diagnosing conditions by\ninterpreting patient data and medical literature with high\nprecision. Another attractive feature of fine-tuning LLMs is\nthat it can be achieved at a cost-efficient manner. While pre-\ntrain", + "type": "text" + }, + { + "rank": 14, + "score": 0.0, + "content": "mon Sense\nMath 14K (MATH) 14K 174 Math\nHellaswag (HE) 10K 272 Common Sense\nGSM8K (GS) 1.3K 148 Math\namount of time [6]. This work uses case study of mathematics\nand common-sense question-answer datasets to demonstrate\nthe fine-tuning process of LLMs.\nB. LoRA\nLow-Rank Adaption (LoRA) is a technique that freezes\nthe pre-trained model weights and injects trainable rank de-\ncomposition into layers of ", + "type": "text" + }, + { + "rank": 15, + "score": 0.0, + "content": "in §III.\nC. Mixture of Experts (MoE)\nThe quality of an LLM is highly related to its scale. Given\na fixed computation budget, it is often desirable to train\na model with more parameters to achieve higher accuracy.\nMixture-of-Experts (MoE) is a technique that, instead of\nusing one large model for all tasks, combines multiple\nexpert sub-networks into a single, large model. As shown\nin Fig. 1, with Mo", + "type": "text" + }, + { + "rank": 16, + "score": 0.0, + "content": "fine-tune two pre-trained MoE models,\nMixtral-8x7B (Mixtral for short) [4] and BlackMamba-\n630M/2.8B (BlackMamba for short) [8]. The details of these\nmodels are shown in Table I. Both models incorporate eight\nexperts in their MoE layers. For dense fine-tuning, all experts\nare activated, whereas for sparse fine-tuning, only the top two\nexperts are selected for each token.\nThese models differ signif", + "type": "text" + }, + { + "rank": 17, + "score": 0.0, + "content": "full BlackMamba model (i.e.,\noriginal weight matrices), whereas employed QLoRA [15]\nfor parameter-efficient fine-tuning (PEFT) on Mixtral due to\nGPU memory capacity budget. For QLoRA, we target the\nMoE layers, including the routers, and set the rank of the\nLoRA modules to 16. We enable FlashAttention2 [17] during\nMixtral fine-tuning for enhanced efficiency. Moreover, we use\ngradient checkpointing ", + "type": "text" + }, + { + "rank": 18, + "score": 0.0, + "content": "ress com-\nmonsense reasoning and arithmetic reasoning respectively\n(provided by LLM-adapters [20]). The details of datasets\nare used in Table II. For evaluation, we tested the models\non GSM8K [21] for arithmetic reasoning and HE [22] for\ncommonsense reasoning. Each dataset consists of thousands\nof queries. We define a query as the concatenation of a\nprompt and its ground-truth answer, which is fee", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "Using\nPyTorch, we provide essential algorithm-level information\nsuch as test accuracy, training throughput, and layer-level\nlatency breakdown. The hardware evaluation offers a detailed\nanalysis of GPU performance. Utilizing NVIDIA Nsight\nCompute [23], we gather kernel-level information, including\nSM utilization, memory utilization, and kernel latency. These\nmetrics collectively offer a comprehensi", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "aset-independent as\nthese workload characteristics do not depend on runtime\ndata. Because profiling is time-consuming (approximately\n10,000 ×costlier compared to a native run without the profiler\nenabled), we manually set the batch size and sequence length\nto facilitate a more direct and efficient profiling process.\nWe present the sequence length distribution for the CS and\nMATH datasets in Fig. 2", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/data/ranked/zschache2025_chunks_ranked.json b/wattbot2025/data/ranked/zschache2025_chunks_ranked.json new file mode 100644 index 00000000..88233043 --- /dev/null +++ b/wattbot2025/data/ranked/zschache2025_chunks_ranked.json @@ -0,0 +1,122 @@ +[ + { + "rank": 1, + "score": 7.414896458551958, + "content": "0\n\"AI development\n(Kaack et al.; Luccioni\net al., 2025). By systematically evaluating\"\n\"inference efficiency and runtime across architectures and hardware settings, we con-\"\n\"tribute to the ongoing discourse on AI’s environmental\nimpact and provide actionable\"\nguidelines for optimizing NLP applications for both performance and sustainability.\n2 Previous research\nResearch on the environmental impac", + "type": "table" + }, + { + "rank": 2, + "score": 6.862966002537272, + "content": "be traced. Our findings have implications\nfor researchers, industry practitioners, and policymakers advocating for sustainable\n2\n\nAI development (Kaack et al.; Luccioni et al., 2025). By systematically evaluating\ninference efficiency and runtime across architectures and hardware settings, we con-\ntribute to the ongoing discourse on AI’s environmental impact and provide actionable\nguidelines for op", + "type": "text" + }, + { + "rank": 3, + "score": 3.540191480736718, + "content": "a system was used. The\"\nminimum number of H100 GPUs required varies by model (see Table B1).\n\"The highest accuracy was achieved by a traditional\nlinear model using pre-trained\"\n\"sentence embeddings. Notably, even the most energy-efficient model\n- a linear model\"\n\"with TF-IDF features - outperformed several\nlarge language models (LLMs). Among\"\n\"LLMs with relatively high accuracy,\nthe best\nsmall mod", + "type": "table" + }, + { + "rank": 4, + "score": 3.4653658158579566, + "content": "model (Qwen 2.5 72B), with only\na minor accuracy reduction of 0.07 points. Deepseek models, despite their extensive\nreasoning processes during inference, exhibit lower accuracy than non-reasoning LLMs\nwhile consuming significantly more energy and taking longer to complete inference.\n4.1 Analysis of hardware settings\nThis section analyzes the impact of different hardware configurations (see Tab. 2)", + "type": "text" + }, + { + "rank": 5, + "score": 3.4597295333805493, + "content": "Zhang, N., Shi, T., Yu, Z., Zhu, M.,\"\n\"Zhang, Y., Song, X., Yang, C., Cheng, Y., Zhao, L.: Beyond Efficiency: A Systematic\"\nSurvey of Resource-Efficient Large Language Models (2024). https://arxiv.org/abs/\n2401.00625\n\"Anthony, L.F.W., Kanding, B., Selvan, R.: Carbontracker: Tracking and Predicting\"\nthe Carbon Footprint of Training Deep Learning Models (2020). https://arxiv.org/\nabs/2007.03051\n\"Hen", + "type": "table" + }, + { + "rank": 6, + "score": 3.4507786496993926, + "content": "and <0.2 dex for both\nenergy consumption and duration (logarithmically scaled to base 10).\nFigure 1 illustrates the trade-off between energy consumption and accuracy across\nall models. For these experiments, a single node of the Capella system was used. The\nminimum number of H100 GPUs required varies by model (see Table B1).\nThe highest accuracy was achieved by a traditional linear model using pre", + "type": "text" + }, + { + "rank": 7, + "score": 3.429773433713179, + "content": "ale models such as GPT-3 (Brown et al.,\n2020), which have significantly advanced task performance. However, this progress\nhas come at a cost: the escalating energy demands of AI systems pose significant\nenvironmental and computational challenges. Data centers that support AI com-\nputations are major electricity consumers, often dependent on fossil fuels, thereby\ncontributing to greenhouse gas emis", + "type": "text" + }, + { + "rank": 8, + "score": 3.414500308516204, + "content": "0\n\"development has been enabled by the Transformer architecture (Vaswani et al., 2017)\"\n\"and exemplified by the emergence of\nlarge-scale models such as GPT-3 (Brown et al.,\"\n\"2020), which have\nsignificantly advanced task performance. However,\nthis progress\"\n\"has\ncome at a cost:\nthe\nescalating energy demands of AI\nsystems pose\nsignificant\"\n\"environmental\nand computational\nchallenges. Data\ncenters\nt", + "type": "table" + }, + { + "rank": 9, + "score": 3.4041343415139367, + "content": "et al. (2019) quantify the carbon footprint\nof NLP models, revealing that the training of a single large-scale transformer model\ncan emit as much carbon as five cars over their entire lifetimes (their measurements\ninclude thousands of hyperparameter tuning jobs, which makes it difficult to disen-\ntangle model-inherent efficiency from experimental setup). This seminal work spurred\nfurther investiga", + "type": "text" + }, + { + "rank": 10, + "score": 3.3932290255404105, + "content": "06.3658542\nBai, G., Chai, Z., Ling, C., Wang, S., Lu, J., Zhang, N., Shi, T., Yu, Z., Zhu, M.,\nZhang, Y., Song, X., Yang, C., Cheng, Y., Zhao, L.: Beyond Efficiency: A Systematic\nSurvey of Resource-Efficient Large Language Models (2024). https://arxiv.org/abs/\n2401.00625\nAnthony, L.F.W., Kanding, B., Selvan, R.: Carbontracker: Tracking and Predicting\nthe Carbon Footprint of Training Deep Learning ", + "type": "text" + }, + { + "rank": 11, + "score": 2.854046563141817, + "content": "0,1,2,3\n\"Luccioni, A.S., Viguier, S., Ligozat, A.-L.: Estimating the carbon footprint of bloom,\",,,\n,a 176b parameter language model. J. Mach. Learn. Res. 24(1) (2023),,\n,\"Gehman, S., Gururangan, S., Sap, M., Choi, Y., Smith, N.A.: Realtoxicityprompts:\",,\n,Evaluating neural toxic degeneration in language models. Proceedings of the 2020,,\nConference,on Empirical Methods,in Natural Language Processi", + "type": "table" + }, + { + "rank": 12, + "score": 2.434261774882816, + "content": "tiative\n\"resulted into the AI Energy Score\n(https://huggingface.co/AIEnergyScore), a tool\"\n\"designed to assess\nthe environmental\nimpact of AI models on a range of\ntasks, and\"\nreinforces the growing importance of considering energy efficiency in model evaluation.\n5.3 Further Requirements of Planet-Centered LLMs\n\"While energy consumption and the associated carbon footprint\nremain crucial con-\"\n\"side", + "type": "table" + }, + { + "rank": 13, + "score": 2.412088041784018, + "content": "Their initiative\nresulted into the AI Energy Score (https://huggingface.co/AIEnergyScore), a tool\ndesigned to assess the environmental impact of AI models on a range of tasks, and\nreinforces the growing importance of considering energy efficiency in model evaluation.\n5.3 Further Requirements of Planet-Centered LLMs\nWhile energy consumption and the associated carbon footprint remain crucial con-\nsi", + "type": "text" + }, + { + "rank": 14, + "score": 2.4011519741118854, + "content": "0\n\"Energy Costs of Large Language Model\nInference\n(2023). https://arxiv.org/abs/\"\n2310.03003\n\"Liu, X., Sun, T., He, J., Wu, J., Wu, L., Zhang, X., Jiang, H., Cao, Z., Huang, X., Qiu,\"\nX.: Towards Efficient NLP: A Standard Evaluation and A Strong Baseline (2022).\nhttps://arxiv.org/abs/2110.07038\n\"Chien, A.A., Lin, L., Nguyen, H., Rao, V., Sharma, T., Wijayawardana, R.: Reducing\"\n\"the carbon impact ", + "type": "table" + }, + { + "rank": 15, + "score": 2.3689307779742474, + "content": "mate carbon footprint when\ntraining deep learning models? a guide and review. Environmental Research\nCommunications 5(11), 115014 (2023) https://doi.org/10.1088/2515-7620/acf81b\nHlavac, M.: stargazer: Well-Formatted Regression and Summary Statistics Tables. R\npackage version 5.2.3 (2022). https://CRAN.R-project.org/package=stargazer\nBrown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariw", + "type": "text" + }, + { + "rank": 16, + "score": 2.3479261408895082, + "content": "mer model\"\ncan emit as much carbon as five cars over their entire lifetimes (their measurements\n\"include thousands of hyperparameter\ntuning jobs, which makes\nit difficult\nto disen-\"\ntangle model-inherent efficiency from experimental setup). This seminal work spurred\n\"further investigations into the environmental costs of training neural networks, includ-\"\n\"ing large language models\n(Patterson et a", + "type": "table" + }, + { + "rank": 17, + "score": 2.3375628874399044, + "content": "Kepner, J., Tiwari, D., Gadepally, V.: From Words to Watts: Benchmarking the\n22\n\nEnergy Costs of Large Language Model Inference (2023). https://arxiv.org/abs/\n2310.03003\nLiu, X., Sun, T., He, J., Wu, J., Wu, L., Zhang, X., Jiang, H., Cao, Z., Huang, X., Qiu,\nX.: Towards Efficient NLP: A Standard Evaluation and A Strong Baseline (2022).\nhttps://arxiv.org/abs/2110.07038\nChien, A.A., Lin, L., Nguyen,", + "type": "text" + }, + { + "rank": 18, + "score": 2.3375628874399044, + "content": "age\nmodels. arXiv preprint arXiv:2302.13971 (2023)\nBender, E.M., Gebru, T., McMillan-Major, A., Shmitchell, S.: On the dangers of\nstochastic parrots: Can language models be too big? Proceedings of the 2021 ACM\nConference on Fairness, Accountability, and Transparency (2021)\n24\n\nLuccioni, A.S., Viguier, S., Ligozat, A.-L.: Estimating the carbon footprint of bloom,\na 176b parameter language model. J.", + "type": "text" + }, + { + "rank": 19, + "score": 0.0, + "content": "Comparing energy consumption and accuracy in\ntext classification inference\nJohannes Zschache and Tilman Hartwig\nApplication Lab for AI and Big Data, German Environment Agency,\nAlte Messe 6, Leipzig, 04103, Saxony, Germany.\n*Corresponding author(s). E-mail(s): tilman.hartwig@uba.de;\nContributing authors: johannes.zschache@uba.de;\nAbstract\nThe increasing deployment of large language models (LLMs) in", + "type": "text" + }, + { + "rank": 20, + "score": 0.0, + "content": "fs between model accuracy and\nenergy consumption in text classification inference across various model archi-\ntectures and hardware configurations. Our empirical analysis shows that the\nbest-performing model in terms of accuracy can also be energy-efficient, while\nlarger LLMs tend to consume significantly more energy with lower classifica-\ntion accuracy. We observe substantial variability in infer", + "type": "text" + } +] \ No newline at end of file diff --git a/wattbot2025/src/retrieval/reranker.py b/wattbot2025/src/retrieval/reranker.py index 0efbc386..92c40123 100644 --- a/wattbot2025/src/retrieval/reranker.py +++ b/wattbot2025/src/retrieval/reranker.py @@ -1,6 +1,47 @@ -"""Reranking module""" +from rank_bm25 import BM25Okapi +import json +from pathlib import Path -class Reranker: - def rerank(self, query, results): - """Rerank search results""" - pass + +class BM25Reranker: + def __init__(self, chunks): + self.chunks = chunks + self.documents = [chunk["content"].split() for chunk in self.chunks] + self.bm25 = BM25Okapi(self.documents) + + def rerank(self, query, top_n=5): + tokenized_query = query.split() + scores = self.bm25.get_scores(tokenized_query) + ranked_indices = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True) + + return [ + { + "rank": i + 1, + "score": float(scores[idx]), + "content": self.chunks[idx]["content"][:400], + "type": self.chunks[idx]["type"] + } + for i, idx in enumerate(ranked_indices[:top_n]) + ] + + +if __name__ == "__main__": + chunks_dir = Path("data/chunks") + ranked_dir = Path("data/ranked") + + ranked_dir.mkdir(parents=True, exist_ok=True) + + query = "carbon emissions reduction goals" + + for file_path in chunks_dir.glob("*.json"): + with open(file_path, "r", encoding="utf-8") as f: + chunks = json.load(f) + + reranker = BM25Reranker(chunks) + results = reranker.rerank(query, top_n=20) + + output_file = ranked_dir / f"{file_path.stem}_ranked.json" + with open(output_file, "w", encoding="utf-8") as f: + json.dump(results, f, indent=2, ensure_ascii=False) + + print(f"Saved ranked results to {output_file}") diff --git a/wattbot2025/tests/test_chunker.py b/wattbot2025/tests/test_chunker.py new file mode 100644 index 00000000..6dd4eb29 --- /dev/null +++ b/wattbot2025/tests/test_chunker.py @@ -0,0 +1,278 @@ +import pytest +import sys +import json +from pathlib import Path +from unittest.mock import Mock, patch, MagicMock + +# Add the src directory to Python path +sys.path.insert(0, str(Path(__file__).parent.parent / "src" / "data")) + +from chunker import DocumentChunker + + +class TestDocumentChunker: + """Comprehensive test suite for DocumentChunker class""" + + @pytest.fixture + def chunker(self): + """Create a chunker instance for testing""" + return DocumentChunker(chunk_size=100, overlap=10) + + @pytest.fixture + def sample_pdf_path(self): + """Path to a real PDF file for testing""" + # Point to papers folder on Desktop + pdf_path = Path.home() / "Desktop" / "papers" + pdfs = list(pdf_path.glob("*.pdf")) + if pdfs: + return str(pdfs[0]) # Return first PDF found + return None + + # ----------------------------- + # TEST CHUNKING LOGIC + # ----------------------------- + + def test_chunk_text_basic(self, chunker): + """Test basic text chunking""" + text = "a" * 250 + chunks = chunker.chunk_text(text) + assert len(chunks) > 1 + assert len(chunks[0]) <= 100 + + def test_chunk_text_overlap(self, chunker): + """Test that chunks have proper overlap""" + text = "0123456789" * 30 + chunks = chunker.chunk_text(text) + assert len(chunks) >= 2 + assert len(chunks[0]) <= 100 + + def test_chunk_text_empty(self, chunker): + """Test chunking empty text""" + chunks = chunker.chunk_text("") + assert len(chunks) == 0 + + def test_chunk_text_shorter_than_chunk_size(self, chunker): + """Test text shorter than chunk_size""" + text = "Short text" + chunks = chunker.chunk_text(text) + assert len(chunks) == 1 + assert chunks[0] == text + + def test_chunk_text_exact_chunk_size(self, chunker): + """Test text exactly equal to chunk_size""" + text = "a" * 100 + chunks = chunker.chunk_text(text) + assert len(chunks) == 2 + + def test_chunk_text_whitespace_handling(self, chunker): + """Test that whitespace is stripped from chunks""" + text = " text with spaces " * 50 + chunks = chunker.chunk_text(text) + for chunk in chunks: + assert chunk == chunk.strip() + + def test_custom_chunk_size(self): + """Test chunker with custom parameters""" + chunker = DocumentChunker(chunk_size=50, overlap=5) + text = "a" * 200 + chunks = chunker.chunk_text(text) + assert len(chunks) > 1 + assert len(chunks[0]) <= 50 + + def test_custom_overlap(self): + """Test chunker with different overlap""" + chunker = DocumentChunker(chunk_size=100, overlap=20) + text = "a" * 300 + chunks = chunker.chunk_text(text) + assert len(chunks) >= 3 + + def test_zero_overlap(self): + """Test chunking with no overlap""" + chunker = DocumentChunker(chunk_size=100, overlap=0) + text = "a" * 300 + chunks = chunker.chunk_text(text) + assert len(chunks) == 3 + + def test_large_overlap(self): + """Test chunking with large overlap (edge case)""" + chunker = DocumentChunker(chunk_size=100, overlap=50) + text = "a" * 300 + chunks = chunker.chunk_text(text) + assert len(chunks) >= 5 + + def test_multiline_text(self, chunker): + """Test chunking text with newlines""" + text = "Line 1\nLine 2\nLine 3\n" * 20 + chunks = chunker.chunk_text(text) + assert len(chunks) > 1 + assert "\n" in chunks[0] + + # ----------------------------- + # TEST PDF TEXT EXTRACTION + # ----------------------------- + + def test_extract_text_from_pdf_real_file(self, chunker, sample_pdf_path): + """Test extracting text from a real PDF file""" + if sample_pdf_path is None: + pytest.skip("No PDF files found for testing") + + text = chunker.extract_text_from_pdf(sample_pdf_path) + assert isinstance(text, str) + assert len(text) > 0 # Should extract some text + + def test_extract_text_from_pdf_returns_string(self, chunker, sample_pdf_path): + """Test that extract_text_from_pdf returns a string""" + if sample_pdf_path is None: + pytest.skip("No PDF files found for testing") + + text = chunker.extract_text_from_pdf(sample_pdf_path) + assert isinstance(text, str) + + @patch("chunker.PdfReader") + def test_extract_text_from_pdf_mock(self, mock_pdf_reader, chunker): + """Test PDF text extraction with mocked PDF""" + # Mock the PDF reader + mock_page = Mock() + mock_page.extract_text.return_value = "Sample text from page" + + mock_reader = Mock() + mock_reader.pages = [mock_page, mock_page] + mock_pdf_reader.return_value = mock_reader + + text = chunker.extract_text_from_pdf("fake.pdf") + assert "Sample text from page" in text + + # ----------------------------- + # TEST TABLE EXTRACTION + # ----------------------------- + + @patch("chunker.camelot.read_pdf") + def test_extract_tables_from_pdf_success(self, mock_camelot, chunker): + """Test successful table extraction""" + # Mock a table + mock_table = Mock() + mock_table.df.to_csv.return_value = "col1,col2\nval1,val2" + + mock_table_list = [mock_table] + mock_camelot.return_value = mock_table_list + + tables = chunker.extract_tables_from_pdf("fake.pdf") + assert len(tables) >= 1 + assert "col1,col2" in tables[0] + + @patch("chunker.camelot.read_pdf") + def test_extract_tables_from_pdf_failure(self, mock_camelot, chunker): + """Test table extraction handles errors gracefully""" + mock_camelot.side_effect = Exception("PDF parsing error") + + tables = chunker.extract_tables_from_pdf("fake.pdf") + # Should return empty list on error (tries both flavors) + assert isinstance(tables, list) + + @patch("chunker.camelot.read_pdf") + def test_extract_tables_both_flavors(self, mock_camelot, chunker): + """Test that both lattice and stream flavors are tried""" + mock_table = Mock() + mock_table.df.to_csv.return_value = "data" + mock_camelot.return_value = [mock_table] + + chunker.extract_tables_from_pdf("fake.pdf") + + # Should be called twice (once for each flavor) + assert mock_camelot.call_count == 2 + + # ----------------------------- + # TEST CHUNK_PDF (INTEGRATION) + # ----------------------------- + + def test_chunk_pdf_real_file(self, chunker, sample_pdf_path): + """Test processing a real PDF file""" + if sample_pdf_path is None: + pytest.skip("No PDF files found for testing") + + chunks = chunker.chunk_pdf(sample_pdf_path) + assert isinstance(chunks, list) + assert len(chunks) > 0 + + # Check structure of chunks + for chunk in chunks: + assert "type" in chunk + assert "content" in chunk + assert chunk["type"] in ["text", "table"] + + @patch.object(DocumentChunker, "extract_text_from_pdf") + @patch.object(DocumentChunker, "extract_tables_from_pdf") + def test_chunk_pdf_creates_text_chunks(self, mock_tables, mock_text, chunker): + """Test that chunk_pdf creates text chunks""" + mock_text.return_value = "Sample text " * 200 + mock_tables.return_value = [] + + chunks = chunker.chunk_pdf("fake.pdf") + + # Should have text chunks + text_chunks = [c for c in chunks if c["type"] == "text"] + assert len(text_chunks) > 0 + + @patch.object(DocumentChunker, "extract_text_from_pdf") + @patch.object(DocumentChunker, "extract_tables_from_pdf") + def test_chunk_pdf_creates_table_chunks(self, mock_tables, mock_text, chunker): + """Test that chunk_pdf creates table chunks""" + mock_text.return_value = "Short text" + mock_tables.return_value = ["col1,col2\n" + "data," * 200] + + chunks = chunker.chunk_pdf("fake.pdf") + + # Should have table chunks + table_chunks = [c for c in chunks if c["type"] == "table"] + assert len(table_chunks) > 0 + + # ----------------------------- + # TEST JSON SAVING + # ----------------------------- + + def test_save_chunks_to_json(self, chunker, tmp_path): + """Test saving chunks to JSON file""" + chunks = [ + {"type": "text", "content": "Sample text"}, + {"type": "table", "content": "col1,col2\nval1,val2"}, + ] + + output_path = tmp_path / "test_output.json" + chunker.save_chunks_to_json(chunks, output_path) + + # Check file was created + assert output_path.exists() + + # Check content is valid JSON + with open(output_path, "r") as f: + loaded_chunks = json.load(f) + + assert len(loaded_chunks) == 2 + assert loaded_chunks[0]["type"] == "text" + assert loaded_chunks[1]["type"] == "table" + + def test_save_chunks_creates_directory(self, chunker, tmp_path): + """Test that save_chunks_to_json creates parent directories""" + output_path = tmp_path / "nested" / "dirs" / "output.json" + chunks = [{"type": "text", "content": "test"}] + + chunker.save_chunks_to_json(chunks, output_path) + + assert output_path.exists() + assert output_path.parent.exists() + + # ----------------------------- + # TEST INITIALIZATION + # ----------------------------- + + def test_init_default_parameters(self): + """Test DocumentChunker initialization with defaults""" + chunker = DocumentChunker() + assert chunker.chunk_size == 1000 + assert chunker.overlap == 100 + + def test_init_custom_parameters(self): + """Test DocumentChunker initialization with custom params""" + chunker = DocumentChunker(chunk_size=500, overlap=50) + assert chunker.chunk_size == 500 + assert chunker.overlap == 50