cargo/sources/registry/index/mod.rs
1//! Management of the index of a registry source.
2//!
3//! This module contains management of the index and various operations, such as
4//! actually parsing the index, looking for crates, etc. This is intended to be
5//! abstract over remote indices (downloaded via Git or HTTP) and local registry
6//! indices (which are all just present on the filesystem).
7//!
8//! ## How the index works
9//!
10//! Here is a simple flow when loading a [`Summary`] (metadata) from the index:
11//!
12//! 1. A query is fired via [`RegistryIndex::query_inner`].
13//! 2. Tries loading all summaries via [`RegistryIndex::load_summaries`], and
14//! under the hood calling [`Summaries::parse`] to parse an index file.
15//! 1. If an on-disk index cache is present, loads it via
16//! [`Summaries::parse_cache`].
17//! 2. Otherwise goes to the slower path [`RegistryData::load`] to get the
18//! specific index file.
19//! 3. A [`Summary`] is now ready in callback `f` in [`RegistryIndex::query_inner`].
20//!
21//! To learn the rationale behind this multi-layer index metadata loading,
22//! see [the documentation of the on-disk index cache](cache).
23use crate::sources::registry::{LoadResponse, RegistryData};
24use crate::util::IntoUrl;
25use crate::util::VersionReqMatchMode;
26use crate::util::data_structures::HashMap;
27use crate::util::interning::InternedString;
28use crate::util::{CargoResult, Filesystem, GlobalContext, OptVersionReq, internal};
29use crate::workspace::dependency::{Artifact, DepKind};
30use crate::workspace::{CliUnstable, Dependency};
31use crate::workspace::{PackageId, SourceId, Summary};
32use cargo_util::registry::make_dep_path;
33use cargo_util_schemas::index::{IndexPackage, RegistryDependency};
34use cargo_util_schemas::manifest::RustVersion;
35use futures::channel::oneshot;
36use semver::Version;
37use serde::{Deserialize, Serialize};
38use std::borrow::Cow;
39use std::cell::RefCell;
40use std::collections::BTreeMap;
41use std::path::Path;
42use std::rc::Rc;
43use std::str;
44use tracing::info;
45
46mod cache;
47use self::cache::CacheManager;
48use self::cache::SummariesCache;
49
50/// The maximum schema version of the `v` field in the index this version of
51/// cargo understands. See [`IndexPackage::v`] for the detail.
52const INDEX_V_MAX: u32 = 2;
53
54/// Manager for handling the on-disk index.
55///
56/// Different kinds of registries store the index differently:
57///
58/// * [`LocalRegistry`] is a simple on-disk tree of files of the raw index.
59/// * [`GitRegistry`] is stored as a raw git repository.
60/// * [`HttpRegistry`] fills the on-disk index cache directly without keeping
61/// any raw index.
62///
63/// These means of access are handled via the [`RegistryData`] trait abstraction.
64/// This transparently handles caching of the index in a more efficient format.
65///
66/// [`LocalRegistry`]: super::local::LocalRegistry
67/// [`GitRegistry`]: super::git_remote::GitRegistry
68/// [`HttpRegistry`]: super::http_remote::HttpRegistry
69pub struct RegistryIndex<'gctx> {
70 source_id: SourceId,
71 /// Root directory of the index for the registry.
72 path: Filesystem,
73 /// In-memory cache of summary data.
74 ///
75 /// This is keyed off the package name. The [`Summaries`] value handles
76 /// loading the summary data. It keeps an optimized on-disk representation
77 /// of the JSON files, which is created in an as-needed fashion. If it
78 /// hasn't been cached already, it uses [`RegistryData::load`] to access
79 /// to JSON files from the index, and the creates the optimized on-disk
80 /// summary cache.
81 summaries_cache: RefCell<HashMap<InternedString, Rc<Summaries>>>,
82 /// Requests that are currently running.
83 summaries_inflight: RefCell<HashMap<InternedString, Vec<oneshot::Sender<Rc<Summaries>>>>>,
84 /// [`GlobalContext`] reference for convenience.
85 gctx: &'gctx GlobalContext,
86 /// Manager of on-disk caches.
87 cache_manager: CacheManager<'gctx>,
88}
89
90/// An internal cache of summaries for a particular package.
91///
92/// A list of summaries are loaded from disk via one of two methods:
93///
94/// 1. From raw registry index --- Primarily Cargo will parse the corresponding
95/// file for a crate in the upstream crates.io registry. That's just a JSON
96/// blob per line which we can parse, extract the version, and then store here.
97/// See [`IndexPackage`] and [`IndexSummary::parse`].
98///
99/// 2. From on-disk index cache --- If Cargo has previously run, we'll have a
100/// cached index of dependencies for the upstream index. This is a file that
101/// Cargo maintains lazily on the local filesystem and is much faster to
102/// parse since it doesn't involve parsing all of the JSON.
103/// See [`SummariesCache`].
104///
105/// The outward-facing interface of this doesn't matter too much where it's
106/// loaded from, but it's important when reading the implementation to note that
107/// we try to parse as little as possible!
108#[derive(Default)]
109struct Summaries {
110 /// A raw vector of uninterpreted bytes. This is what `Unparsed` start/end
111 /// fields are indexes into. If a `Summaries` is loaded from the crates.io
112 /// index then this field will be empty since nothing is `Unparsed`.
113 raw_data: Vec<u8>,
114
115 /// All known versions of a crate, keyed from their `Version` to the
116 /// possibly parsed or unparsed version of the full summary.
117 versions: Vec<(Version, RefCell<MaybeIndexSummary>)>,
118}
119
120/// A lazily parsed [`IndexSummary`].
121enum MaybeIndexSummary {
122 /// A summary which has not been parsed, The range are pointers
123 /// into [`Summaries::raw_data`] which this is an entry of.
124 Unparsed(std::range::Range<usize>),
125
126 /// An actually parsed summary.
127 Parsed(IndexSummary),
128}
129
130/// A parsed representation of a summary from the index. This is usually parsed
131/// from a line from a raw index file, or a JSON blob from on-disk index cache.
132///
133/// In addition to a full [`Summary`], we have information on whether it is `yanked`.
134#[derive(Clone, Debug, PartialEq, Eq, Hash)]
135pub enum IndexSummary {
136 /// Available for consideration
137 Candidate(Summary),
138 /// Yanked within its registry
139 Yanked(Summary),
140 /// Not available as we are offline and create is not downloaded yet
141 Offline(Summary),
142 /// From a newer schema version and is likely incomplete or inaccurate
143 Unsupported(Summary, u32),
144 /// An error was encountered despite being a supported schema version
145 Invalid(Summary),
146}
147
148impl IndexSummary {
149 /// Extract the summary from any variant.
150 ///
151 /// You should not use this unless you know what you are doing.
152 fn as_summary_unchecked(&self) -> &Summary {
153 match self {
154 IndexSummary::Candidate(sum)
155 | IndexSummary::Yanked(sum)
156 | IndexSummary::Offline(sum)
157 | IndexSummary::Unsupported(sum, _)
158 | IndexSummary::Invalid(sum) => sum,
159 }
160 }
161
162 pub fn map_summary(self, f: impl Fn(Summary) -> Summary) -> Self {
163 match self {
164 IndexSummary::Candidate(s) => IndexSummary::Candidate(f(s)),
165 IndexSummary::Yanked(s) => IndexSummary::Yanked(f(s)),
166 IndexSummary::Offline(s) => IndexSummary::Offline(f(s)),
167 IndexSummary::Unsupported(s, v) => IndexSummary::Unsupported(f(s), v.clone()),
168 IndexSummary::Invalid(s) => IndexSummary::Invalid(f(s)),
169 }
170 }
171
172 /// Extract the package id from any variant
173 pub fn package_id(&self) -> PackageId {
174 self.as_summary_unchecked().package_id()
175 }
176
177 /// Returns `true` if the index summary is [`Yanked`].
178 ///
179 /// [`Yanked`]: IndexSummary::Yanked
180 #[must_use]
181 pub fn is_yanked(&self) -> bool {
182 matches!(self, Self::Yanked(..))
183 }
184
185 /// Returns `true` if the index summary is [`Offline`].
186 ///
187 /// [`Offline`]: IndexSummary::Offline
188 #[must_use]
189 pub fn is_offline(&self) -> bool {
190 matches!(self, Self::Offline(..))
191 }
192}
193
194fn index_package_to_summary(
195 pkg: IndexPackage<'_>,
196 source_id: SourceId,
197 cli_unstable: &CliUnstable,
198) -> CargoResult<Summary> {
199 // ****CAUTION**** Please be extremely careful with returning errors, see
200 // `IndexSummary::parse` for details
201 let pkgid = PackageId::new(pkg.name.as_ref().into(), pkg.vers, source_id);
202 let deps = pkg
203 .deps
204 .into_iter()
205 .map(|dep| registry_dependency_into_dep(dep, source_id, cli_unstable))
206 .collect::<CargoResult<Vec<_>>>()?;
207 let mut features = pkg.features;
208 if let Some(features2) = pkg.features2 {
209 for (name, values) in features2 {
210 features.entry(name).or_default().extend(values);
211 }
212 }
213 let features = features
214 .into_iter()
215 .map(|(name, values)| (name.into(), values.into_iter().map(|v| v.into()).collect()))
216 .collect::<BTreeMap<_, _>>();
217 let links: Option<InternedString> = pkg.links.as_ref().map(|l| l.as_ref().into());
218 let mut summary = Summary::new(pkgid, deps, &features, links, pkg.rust_version)?;
219 summary.set_checksum(pkg.cksum);
220 if let Some(pubtime) = pkg.pubtime {
221 summary.set_pubtime(pubtime);
222 }
223 Ok(summary)
224}
225
226#[derive(Deserialize, Serialize)]
227struct IndexPackageMinimum<'a> {
228 name: Cow<'a, str>,
229 vers: Version,
230}
231
232#[derive(Deserialize, Serialize, Default)]
233struct IndexPackageRustVersion {
234 rust_version: Option<RustVersion>,
235}
236
237#[derive(Deserialize, Serialize, Default)]
238struct IndexPackageV {
239 v: Option<u32>,
240}
241
242impl<'gctx> RegistryIndex<'gctx> {
243 /// Creates an empty registry index at `path`.
244 pub fn new(
245 source_id: SourceId,
246 path: &Filesystem,
247 gctx: &'gctx GlobalContext,
248 ) -> RegistryIndex<'gctx> {
249 RegistryIndex {
250 source_id,
251 path: path.clone(),
252 summaries_cache: RefCell::new(HashMap::default()),
253 summaries_inflight: RefCell::new(HashMap::default()),
254 gctx,
255 cache_manager: CacheManager::new(path.join(".cache"), gctx),
256 }
257 }
258
259 /// Returns the hash listed for a specified `PackageId`. Primarily for
260 /// checking the integrity of a downloaded package matching the checksum in
261 /// the index file, aka [`IndexSummary`].
262 pub async fn hash(&self, pkg: PackageId, load: &dyn RegistryData) -> CargoResult<String> {
263 let req = OptVersionReq::lock_to_exact(pkg.version());
264 let mut summary = self.summaries(pkg.name(), &req, load).await?;
265 Ok(summary
266 .next()
267 .ok_or_else(|| internal(format!("no hash listed for {}", pkg)))?
268 .as_summary_unchecked()
269 .checksum()
270 .map(|checksum| checksum.to_string())
271 .ok_or_else(|| internal(format!("no hash listed for {}", pkg)))?)
272 }
273
274 /// Load a list of summaries for `name` package in this registry which
275 /// match `req`.
276 ///
277 /// This function will semantically
278 ///
279 /// 1. parse the index file (either raw or cache),
280 /// 2. match all versions,
281 /// 3. and then return an iterator over all summaries which matched.
282 ///
283 /// Internally there's quite a few layer of caching to amortize this cost
284 /// though since this method is called quite a lot on null builds in Cargo.
285 async fn summaries<'a, 'b>(
286 &'a self,
287 name: InternedString,
288 req: &'b OptVersionReq,
289 load: &dyn RegistryData,
290 ) -> CargoResult<impl Iterator<Item = IndexSummary> + 'b>
291 where
292 'a: 'b,
293 {
294 // First up parse what summaries we have available.
295 let summaries = self.load_summaries(name, load).await?;
296
297 // Iterate over our summaries, extract all relevant ones which match our
298 // version requirement, and then parse all corresponding rows in the
299 // registry. As a reminder this `summaries` method is called for each
300 // entry in a lock file on every build, so we want to absolutely
301 // minimize the amount of work being done here and parse as little as
302 // necessary.
303
304 struct I<'a> {
305 name: InternedString,
306 index: &'a RegistryIndex<'a>,
307 req: &'a OptVersionReq,
308 summaries: Rc<Summaries>,
309 i: usize,
310 }
311
312 impl<'a> Iterator for I<'a> {
313 type Item = IndexSummary;
314
315 fn next(&mut self) -> Option<Self::Item> {
316 while let Some((v, summary)) = self.summaries.versions.get(self.i) {
317 self.i += 1;
318 if self.req.matches(v, VersionReqMatchMode::Default) {
319 match summary.borrow_mut().parse(
320 &self.summaries.raw_data,
321 self.index.source_id,
322 self.index.gctx.cli_unstable(),
323 ) {
324 Ok(summary) => return Some(summary.clone()),
325 Err(e) => {
326 info!("failed to parse `{}` registry package: {}", self.name, e);
327 }
328 }
329 }
330 }
331 None
332 }
333 }
334
335 Ok(I {
336 name,
337 index: self,
338 req,
339 summaries,
340 i: 0,
341 })
342 }
343
344 /// Actually parses what summaries we have available.
345 ///
346 /// If Cargo has run previously, this tries in this order:
347 ///
348 /// 1. Returns from in-memory cache, aka [`RegistryIndex::summaries_cache`].
349 /// 2. If missing, hands over to [`Summaries::parse`] to parse an index file.
350 ///
351 /// The actual kind index file being parsed depends on which kind of
352 /// [`RegistryData`] the `load` argument is given. For example, a
353 /// Git-based [`GitRegistry`] will first try a on-disk index cache
354 /// file, and then try parsing registry raw index from Git repository.
355 ///
356 /// In effect, this is intended to be a quite cheap operation.
357 ///
358 /// [`GitRegistry`]: super::git_remote::GitRegistry
359 async fn load_summaries(
360 &self,
361 name: InternedString,
362 load: &dyn RegistryData,
363 ) -> CargoResult<Rc<Summaries>> {
364 // If we've previously loaded what versions are present for `name`, just
365 // return that since our in-memory cache should still be valid.
366 if let Some(summaries) = self.summaries_cache.borrow().get(&name) {
367 return Ok(summaries.clone());
368 }
369
370 // Check if this request has already started. If so, return a oneshot that hands out the same data.
371 let rx = {
372 let mut pending = self.summaries_inflight.borrow_mut();
373 if let Some(waiters) = pending.get_mut(&name) {
374 let (tx, rx) = oneshot::channel();
375 waiters.push(tx);
376 Some(rx)
377 } else {
378 // We'll be the one to do the work. When we're done, we'll let all the pending queries know.
379 pending.insert(name, Vec::new());
380 None
381 }
382 };
383 if let Some(rx) = rx {
384 return Ok(rx.await?);
385 }
386
387 let summaries = self.load_summaries_uncached(name, load).await;
388 let pending = self.summaries_inflight.borrow_mut().remove(&name).unwrap();
389 if let Ok(summaries) = &summaries {
390 // Insert into the cache
391 self.summaries_cache
392 .borrow_mut()
393 .insert(name, summaries.clone());
394
395 // Send the value to all waiting futures.
396 for entry in pending {
397 let _ = entry.send(summaries.clone());
398 }
399 };
400 summaries
401 }
402
403 async fn load_summaries_uncached(
404 &self,
405 name: InternedString,
406 load: &dyn RegistryData,
407 ) -> CargoResult<Rc<Summaries>> {
408 // Prepare the `RegistryData` which will lazily initialize internal data
409 // structures.
410 load.prepare()?;
411
412 let root = load.assert_index_locked(&self.path);
413 let summaries = Summaries::parse(
414 root,
415 &name,
416 self.source_id,
417 load,
418 self.gctx.cli_unstable(),
419 &self.cache_manager,
420 )
421 .await?
422 .unwrap_or_default();
423 Ok(Rc::new(summaries))
424 }
425
426 /// Clears the in-memory summaries cache.
427 pub fn clear_summaries_cache(&self) {
428 self.summaries_cache.borrow_mut().clear();
429 }
430
431 pub async fn query_inner(
432 &self,
433 name: InternedString,
434 req: &OptVersionReq,
435 load: &dyn RegistryData,
436 f: &mut dyn FnMut(IndexSummary),
437 ) -> CargoResult<()> {
438 if !self.gctx.network_allowed() {
439 // This should only return `Ok(())` if there is at least 1 match.
440 //
441 // If there are 0 matches it should fall through and try again with online.
442 // This is necessary for dependencies that are not used (such as
443 // target-cfg or optional), but are not downloaded. Normally the
444 // build should succeed if they are not downloaded and not used,
445 // but they still need to resolve. If they are actually needed
446 // then cargo will fail to download and an error message
447 // indicating that the required dependency is unavailable while
448 // offline will be displayed.
449 let mut called = false;
450 let callback = &mut |s: IndexSummary| {
451 if !s.is_offline() {
452 called = true;
453 f(s);
454 }
455 };
456 self.query_inner_with_online(name, req, load, callback, false)
457 .await?;
458 if called {
459 return Ok(());
460 }
461 }
462 self.query_inner_with_online(name, req, load, f, true).await
463 }
464
465 /// Inner implementation of [`Self::query_inner`]. Returns the number of
466 /// summaries we've got.
467 ///
468 /// The `online` controls whether Cargo can access the network when needed.
469 async fn query_inner_with_online(
470 &self,
471 name: InternedString,
472 req: &OptVersionReq,
473 load: &dyn RegistryData,
474 f: &mut dyn FnMut(IndexSummary),
475 online: bool,
476 ) -> CargoResult<()> {
477 self.summaries(name, &req, load)
478 .await?
479 // First filter summaries for `--offline`. If we're online then
480 // everything is a candidate, otherwise if we're offline we're only
481 // going to consider candidates which are actually present on disk.
482 //
483 // Note: This particular logic can cause problems with
484 // optional dependencies when offline. If at least 1 version
485 // of an optional dependency is downloaded, but that version
486 // does not satisfy the requirements, then resolution will
487 // fail. Unfortunately, whether or not something is optional
488 // is not known here.
489 .map(|s| {
490 if online || load.is_crate_downloaded(s.package_id()) {
491 s.clone()
492 } else {
493 IndexSummary::Offline(s.as_summary_unchecked().clone())
494 }
495 })
496 .for_each(f);
497 Ok(())
498 }
499}
500
501impl Summaries {
502 /// Parse out a [`Summaries`] instances from on-disk state.
503 ///
504 /// This will do the followings in order:
505 ///
506 /// 1. Attempt to prefer parsing a previous index cache file that already
507 /// exists from a previous invocation of Cargo (aka you're typing `cargo
508 /// build` again after typing it previously).
509 /// 2. If parsing fails, or the cache isn't found or is invalid, we then
510 /// take a slower path which loads the full descriptor for `relative`
511 /// from the underlying index (aka libgit2 with crates.io, or from a
512 /// remote HTTP index) and then parse everything in there.
513 ///
514 /// * `root` --- this is the root argument passed to `load`
515 /// * `name` --- the name of the package.
516 /// * `source_id` --- the registry's `SourceId` used when parsing JSON blobs
517 /// to create summaries.
518 /// * `load` --- the actual index implementation which may be very slow to
519 /// call. We avoid this if we can.
520 /// * `bindeps` --- whether the `-Zbindeps` unstable flag is enabled
521 pub async fn parse(
522 root: &Path,
523 name: &str,
524 source_id: SourceId,
525 load: &dyn RegistryData,
526 cli_unstable: &CliUnstable,
527 cache_manager: &CacheManager<'_>,
528 ) -> CargoResult<Option<Summaries>> {
529 // This is the file we're loading from cache or the index data.
530 // See module comment in `registry/mod.rs` for why this is structured the way it is.
531 let lowered_name = &name.to_lowercase();
532 let relative = make_dep_path(&lowered_name, false);
533
534 let mut cached_summaries = None;
535 let mut index_version = None;
536 if let Some(contents) = cache_manager.get(lowered_name) {
537 match Summaries::parse_cache(contents) {
538 Ok((s, v)) => {
539 cached_summaries = Some(s);
540 index_version = Some(v);
541 }
542 Err(e) => {
543 tracing::debug!("failed to parse {lowered_name:?} cache: {e}");
544 }
545 }
546 }
547
548 let response = load
549 .load(root, relative.as_ref(), index_version.as_deref())
550 .await?;
551
552 match response {
553 LoadResponse::CacheValid => {
554 tracing::debug!("fast path for registry cache of {:?}", relative);
555 if cached_summaries.is_none() {
556 return Err(anyhow::anyhow!(
557 "registry said cache valid when no cache exists"
558 ));
559 }
560 return Ok(cached_summaries);
561 }
562 LoadResponse::NotFound => {
563 cache_manager.invalidate(lowered_name);
564 return Ok(None);
565 }
566 LoadResponse::Data {
567 raw_data,
568 index_version,
569 } => {
570 // This is the fallback path where we actually talk to the registry backend to load
571 // information. Here we parse every single line in the index (as we need
572 // to find the versions)
573 tracing::debug!("slow path for {:?}", relative);
574 let mut cache = SummariesCache::default();
575 let mut ret = Summaries::default();
576 ret.raw_data = raw_data;
577 for line in split(&ret.raw_data, b'\n') {
578 // Attempt forwards-compatibility on the index by ignoring
579 // everything that we ourselves don't understand, that should
580 // allow future cargo implementations to break the
581 // interpretation of each line here and older cargo will simply
582 // ignore the new lines.
583 let summary = match IndexSummary::parse(line, source_id, cli_unstable) {
584 Ok(summary) => summary,
585 Err(e) => {
586 // This should only happen when there is an index
587 // entry from a future version of cargo that this
588 // version doesn't understand. Hopefully, those future
589 // versions of cargo correctly set INDEX_V_MAX and
590 // CURRENT_CACHE_VERSION, otherwise this will skip
591 // entries in the cache preventing those newer
592 // versions from reading them (that is, until the
593 // cache is rebuilt).
594 tracing::info!(
595 "failed to parse {:?} registry package: {}",
596 relative,
597 e
598 );
599 continue;
600 }
601 };
602 let version = summary.package_id().version().clone();
603 cache.versions.push((version.clone(), line));
604 ret.versions.push((version, RefCell::new(summary.into())));
605 }
606 if let Some(index_version) = index_version {
607 tracing::trace!("caching index_version {}", index_version);
608 let cache_bytes = cache.serialize(index_version.as_str());
609 // Once we have our `cache_bytes` which represents the `Summaries` we're
610 // about to return, write that back out to disk so future Cargo
611 // invocations can use it.
612 cache_manager.put(lowered_name, &cache_bytes);
613
614 // If we've got debug assertions enabled read back in the cached values
615 // and assert they match the expected result.
616 #[cfg(debug_assertions)]
617 {
618 let readback = SummariesCache::parse(&cache_bytes)
619 .expect("failed to parse cache we just wrote");
620 assert_eq!(
621 readback.index_version, index_version,
622 "index_version mismatch"
623 );
624 assert_eq!(readback.versions, cache.versions, "versions mismatch");
625 }
626 }
627 Ok(Some(ret))
628 }
629 }
630 }
631
632 /// Parses the contents of an on-disk cache, aka [`SummariesCache`], which
633 /// represents information previously cached by Cargo.
634 pub fn parse_cache(contents: Vec<u8>) -> CargoResult<(Summaries, InternedString)> {
635 let cache = SummariesCache::parse(&contents)?;
636 let index_version = cache.index_version.into();
637 let mut versions = Vec::with_capacity(cache.versions.len());
638 for (version, summary) in cache.versions {
639 let range = contents.subslice_range(summary).unwrap();
640 versions.push((version, RefCell::new(MaybeIndexSummary::Unparsed(range))));
641 }
642 let ret = Summaries {
643 raw_data: contents,
644 versions,
645 };
646
647 Ok((ret, index_version))
648 }
649}
650
651impl MaybeIndexSummary {
652 /// Parses this "maybe a summary" into a `Parsed` for sure variant.
653 ///
654 /// Does nothing if this is already `Parsed`, and otherwise the `raw_data`
655 /// passed in is sliced with the bounds in `Unparsed` and then actually
656 /// parsed.
657 fn parse(
658 &mut self,
659 raw_data: &[u8],
660 source_id: SourceId,
661 cli_unstable: &CliUnstable,
662 ) -> CargoResult<&IndexSummary> {
663 let range = match self {
664 MaybeIndexSummary::Unparsed(range) => *range,
665 MaybeIndexSummary::Parsed(summary) => return Ok(summary),
666 };
667 let summary = IndexSummary::parse(&raw_data[range], source_id, cli_unstable)?;
668 *self = MaybeIndexSummary::Parsed(summary);
669 match self {
670 MaybeIndexSummary::Unparsed(_) => unreachable!(),
671 MaybeIndexSummary::Parsed(summary) => Ok(summary),
672 }
673 }
674}
675
676impl From<IndexSummary> for MaybeIndexSummary {
677 fn from(summary: IndexSummary) -> MaybeIndexSummary {
678 MaybeIndexSummary::Parsed(summary)
679 }
680}
681
682impl IndexSummary {
683 /// Parses a line from the registry's index file into an [`IndexSummary`]
684 /// for a package.
685 ///
686 /// The `line` provided is expected to be valid JSON. It is supposed to be
687 /// a [`IndexPackage`].
688 fn parse(
689 line: &[u8],
690 source_id: SourceId,
691 cli_unstable: &CliUnstable,
692 ) -> CargoResult<IndexSummary> {
693 // Subset of the index that is used at the end of the function
694 // Extracted so that we do not have to clone IndexPackage's fields
695 struct IndexSubset {
696 v: Option<u32>,
697 yanked: Option<bool>,
698 }
699
700 // ****CAUTION**** Please be extremely careful with returning errors
701 // from this function. Entries that error are not included in the
702 // index cache, and can cause cargo to get confused when switching
703 // between different versions that understand the index differently.
704 // Make sure to consider the INDEX_V_MAX and CURRENT_CACHE_VERSION
705 // values carefully when making changes here.
706 let index_summary = (|| {
707 let index = serde_json::from_slice::<IndexPackage<'_>>(line)?;
708 let subset = IndexSubset {
709 v: index.v,
710 yanked: index.yanked,
711 };
712 tracing::trace!("json parsed registry {}/{}", index.name, index.vers);
713 let summary = index_package_to_summary(index, source_id, cli_unstable)?;
714 Ok((subset, summary))
715 })();
716 let (subset, summary, valid) = match index_summary {
717 Ok((subset, summary)) => (subset, summary, true),
718 Err(err) => {
719 let Ok(IndexPackageMinimum { name, vers }) =
720 serde_json::from_slice::<IndexPackageMinimum<'_>>(line)
721 else {
722 // If we can't recover, prefer the original error
723 return Err(err);
724 };
725 tracing::info!(
726 "recoverying from failed parse of registry package {name}@{vers}: {err}"
727 );
728 let IndexPackageRustVersion { rust_version } =
729 serde_json::from_slice::<IndexPackageRustVersion>(line).unwrap_or_default();
730 let IndexPackageV { v } =
731 serde_json::from_slice::<IndexPackageV>(line).unwrap_or_default();
732 let subset = IndexSubset {
733 v,
734 yanked: Default::default(),
735 };
736 let index = IndexPackage {
737 name,
738 vers,
739 rust_version,
740 v,
741 deps: Default::default(),
742 features: Default::default(),
743 features2: Default::default(),
744 cksum: Default::default(),
745 yanked: Default::default(),
746 links: Default::default(),
747 pubtime: Default::default(),
748 };
749 tracing::trace!("json parsed registry {}/{}", index.name, index.vers);
750 let summary = index_package_to_summary(index, source_id, cli_unstable)?;
751 (subset, summary, false)
752 }
753 };
754 let v = subset.v.unwrap_or(1);
755
756 let v_max = if cli_unstable.bindeps {
757 INDEX_V_MAX + 1
758 } else {
759 INDEX_V_MAX
760 };
761
762 if v_max < v {
763 Ok(IndexSummary::Unsupported(summary, v))
764 } else if !valid {
765 Ok(IndexSummary::Invalid(summary))
766 } else if subset.yanked.unwrap_or(false) {
767 Ok(IndexSummary::Yanked(summary))
768 } else {
769 Ok(IndexSummary::Candidate(summary))
770 }
771 }
772}
773
774/// Converts an encoded dependency in the registry to a cargo dependency
775fn registry_dependency_into_dep(
776 dep: RegistryDependency<'_>,
777 default: SourceId,
778 cli_unstable: &CliUnstable,
779) -> CargoResult<Dependency> {
780 let RegistryDependency {
781 name,
782 req,
783 mut features,
784 optional,
785 default_features,
786 target,
787 kind,
788 registry,
789 package,
790 public,
791 artifact,
792 bindep_target,
793 lib,
794 } = dep;
795
796 let id = if let Some(registry) = ®istry {
797 SourceId::for_registry(®istry.into_url()?)?
798 } else {
799 default
800 };
801
802 let interned_name = InternedString::new(package.as_ref().unwrap_or(&name));
803 let mut dep = Dependency::parse(interned_name, Some(&req), id)?;
804 if package.is_some() {
805 dep.set_explicit_name_in_toml(name);
806 }
807 let kind = match kind.as_deref().unwrap_or("") {
808 "dev" => DepKind::Development,
809 "build" => DepKind::Build,
810 _ => DepKind::Normal,
811 };
812
813 let platform = match target {
814 Some(target) => Some(target.parse()?),
815 None => None,
816 };
817
818 // All dependencies are private by default
819 let public = public.unwrap_or(false);
820
821 // Unfortunately older versions of cargo and/or the registry ended up
822 // publishing lots of entries where the features array contained the
823 // empty feature, "", inside. This confuses the resolution process much
824 // later on and these features aren't actually valid, so filter them all
825 // out here.
826 features.retain(|s| !s.is_empty());
827
828 // In index, "registry" is null if it is from the same index.
829 // In Cargo.toml, "registry" is None if it is from the default
830 if !id.is_crates_io() {
831 dep.set_registry_id(id);
832 }
833
834 if let Some(artifacts) = artifact {
835 let artifact = Artifact::parse(
836 &artifacts,
837 lib,
838 bindep_target.as_deref(),
839 cli_unstable.json_target_spec,
840 )?;
841 dep.set_artifact(artifact);
842 }
843
844 dep.set_optional(optional)
845 .set_default_features(default_features)
846 .set_features(features)
847 .set_platform(platform)
848 .set_kind(kind)
849 .set_public(public);
850
851 Ok(dep)
852}
853
854/// Like [`slice::split`] but is optimized by [`memchr`].
855fn split(haystack: &[u8], needle: u8) -> impl Iterator<Item = &[u8]> {
856 struct Split<'a> {
857 haystack: &'a [u8],
858 needle: u8,
859 }
860
861 impl<'a> Iterator for Split<'a> {
862 type Item = &'a [u8];
863
864 fn next(&mut self) -> Option<&'a [u8]> {
865 if self.haystack.is_empty() {
866 return None;
867 }
868 let (ret, remaining) = match memchr::memchr(self.needle, self.haystack) {
869 Some(pos) => (&self.haystack[..pos], &self.haystack[pos + 1..]),
870 None => (self.haystack, &[][..]),
871 };
872 self.haystack = remaining;
873 Some(ret)
874 }
875 }
876
877 Split { haystack, needle }
878}