@@ -118,6 +118,19 @@ fn strip_query_and_fragment(url: &str) -> &str {
118118 }
119119}
120120
121+ /// The file a directory-relative link names, resolved against the directory
122+ /// holding the document that wrote it.
123+ ///
124+ /// The index keeps a destination as the document spelled it, because that is the
125+ /// text an edit to the link has to be measured against, and a spelling can carry
126+ /// a query string. A query is not part of a file name - no file is ever called
127+ /// `b.md?raw=true` - so it is stripped here. Every consumer asking which file a
128+ /// link points at goes through this, so the index's own keys and the answers
129+ /// navigation gives cannot disagree.
130+ pub fn link_target_file ( source_dir : & Path , target_path : & str ) -> PathBuf {
131+ WorkspaceIndex :: normalize_path ( & source_dir. join ( strip_query_and_fragment ( target_path) ) )
132+ }
133+
121134/// Markdown file links extracted from a document, split by how they resolve.
122135///
123136/// Linting rules only understand `relative` links (resolved against the source
@@ -270,8 +283,14 @@ const CACHE_MAGIC: &[u8; 4] = b"RWSI";
270283/// no longer correct. Version 9 adds `CrossFileLinkIndex::origin`; postcard is
271284/// not self-describing, so a version 8 cache would decode the following field's
272285/// bytes as the new one and yield nonsense.
286+ ///
287+ /// Version 10 changes what `cross_file_links` holds rather than how it is laid
288+ /// out: one link is now one entry however each rule spells the destination. The
289+ /// bytes still decode, so nothing here would notice, and a cached index is
290+ /// reused whole when a file's content is unchanged - a version 9 cache would
291+ /// keep reporting the duplicate this version exists to stop.
273292#[ cfg( feature = "postcard" ) ]
274- const CACHE_FORMAT_VERSION : u32 = 9 ;
293+ const CACHE_FORMAT_VERSION : u32 = 10 ;
275294
276295/// Cache file name within the version directory
277296#[ cfg( feature = "postcard" ) ]
@@ -757,15 +776,13 @@ impl WorkspaceIndex {
757776 }
758777
759778 /// Resolve a relative path from a source file to an absolute target path
779+ ///
780+ /// This keys the reverse dependency graph, which is looked up by the path of
781+ /// a file that changed, so it has to answer with a file name - the same
782+ /// question [`link_target_file`] answers for every other consumer.
760783 fn resolve_target_path ( & self , source_file : & Path , relative_target : & str ) -> PathBuf {
761- // Get the directory containing the source file
762784 let source_dir = source_file. parent ( ) . unwrap_or ( Path :: new ( "" ) ) ;
763-
764- // Join with the relative target and normalize
765- let class=pl-kos>.join ( relative_target) ;
766-
767- // Normalize the path (handle .., ., etc.)
768- Self :: normalize_path ( & target)
785+ link_target_file ( source_dir, relative_target)
769786 }
770787
771788 /// Normalize a path by resolving . and .. components
@@ -964,15 +981,35 @@ impl FileIndex {
964981 false
965982 }
966983
967- /// Add a cross-file link to the index (deduplicates by target_path, fragment, line)
984+ /// Add a cross-file link to the index, keyed on the file it names, the
985+ /// fragment it asks for, and the line it sits on.
986+ ///
987+ /// Several rules contribute the same link and spell it differently: MD051
988+ /// records the destination as written and starts at the link, MD057 records
989+ /// the file that destination names and starts at the URL. Neither the string
990+ /// nor the column can identify a link, so the file it names does - comparing
991+ /// the raw strings let `page.md?raw=true` in as a second entry alongside
992+ /// `page.md`, and MD051, which reports every entry, then reported the same
993+ /// broken fragment twice.
994+ ///
995+ /// One key per line is what the index has always recorded, so two links on
996+ /// one line asking the same file for the same fragment are one entry
997+ /// however each of them spells the destination.
968998 pub fn add_cross_file_link ( & mut self , link : CrossFileLinkIndex ) {
969- // Deduplicate: multiple rules may contribute the same link with different columns
970- // (e.g., MD051 uses link start, MD057 uses URL start)
971- let is_duplicate = self . cross_file_links . iter ( ) . any ( | existing| {
972- existing . target_path == link . target_path && existing . fragment == link . fragment && existing. line == link. line
999+ let existing = self . cross_file_links . iter_mut ( ) . find ( |existing| {
1000+ existing . fragment == link . fragment
1001+ && existing. line == link . line
1002+ && strip_query_and_fragment ( & existing. target_path ) == strip_query_and_fragment ( & link. target_path )
9731003 } ) ;
974- if !is_duplicate {
975- self . cross_file_links . push ( link) ;
1004+ match existing {
1005+ // A message quotes the destination back, so the spelling that kept the
1006+ // query string is the one to keep, whichever rule recorded it first.
1007+ Some ( existing) => {
1008+ if !existing. target_path . contains ( '?' ) && link. target_path . contains ( '?' ) {
1009+ * existing = link;
1010+ }
1011+ }
1012+ None => self . cross_file_links . push ( link) ,
9761013 }
9771014 }
9781015
@@ -1175,6 +1212,125 @@ mod tests {
11751212 ) ;
11761213 }
11771214
1215+ /// Two rules record the same link, one keeping the query string and one
1216+ /// keeping only the file it names. That is one link, and the spelling kept
1217+ /// is the destination as written whichever rule got there first - a message
1218+ /// quotes it back, so the answer must not depend on rule order.
1219+ #[ test]
1220+ fn test_add_cross_file_link_keeps_the_destination_as_written ( ) {
1221+ let as_written = CrossFileLinkIndex {
1222+ target_path : "other.md?raw=true" . to_string ( ) ,
1223+ fragment : "missing" . to_string ( ) ,
1224+ line : 3 ,
1225+ column : 1 ,
1226+ origin : LinkOrigin :: Body ,
1227+ } ;
1228+ let file_named = CrossFileLinkIndex {
1229+ target_path : "other.md" . to_string ( ) ,
1230+ fragment : "missing" . to_string ( ) ,
1231+ line : 3 ,
1232+ column : 9 ,
1233+ origin : LinkOrigin :: Body ,
1234+ } ;
1235+
1236+ for ( first, second) in [
1237+ ( as_written. clone ( ) , file_named. clone ( ) ) ,
1238+ ( file_named. clone ( ) , as_written. clone ( ) ) ,
1239+ ] {
1240+ let mut index = FileIndex :: new ( ) ;
1241+ index. add_cross_file_link ( first) ;
1242+ index. add_cross_file_link ( second) ;
1243+
1244+ assert_eq ! (
1245+ index. cross_file_links. len( ) ,
1246+ 1 ,
1247+ "one link is one entry, got: {:?}" ,
1248+ index. cross_file_links
1249+ ) ;
1250+ assert_eq ! ( index. cross_file_links[ 0 ] . target_path, "other.md?raw=true" ) ;
1251+ }
1252+ }
1253+
1254+ /// Links to two different files are two entries, so the deduplication above
1255+ /// cannot swallow a second target.
1256+ #[ test]
1257+ fn test_add_cross_file_link_keeps_distinct_targets ( ) {
1258+ let mut index = FileIndex :: new ( ) ;
1259+ for target in [ "one.md" , "two.md" ] {
1260+ index. add_cross_file_link ( CrossFileLinkIndex {
1261+ target_path : target. to_string ( ) ,
1262+ fragment : "missing" . to_string ( ) ,
1263+ line : 3 ,
1264+ column : 1 ,
1265+ origin : LinkOrigin :: Body ,
1266+ } ) ;
1267+ }
1268+ assert_eq ! ( index. cross_file_links. len( ) , 2 ) ;
1269+ }
1270+
1271+ /// Two links on one line asking the same file for the same fragment are one
1272+ /// entry, which is what the index has always recorded for two identically
1273+ /// spelled destinations. Differing query strings do not make them two links,
1274+ /// because a query string is not part of a file name.
1275+ ///
1276+ /// A different fragment, or the same link on another line, stays its own
1277+ /// entry - so this is a boundary, not a blanket collapse to one finding.
1278+ #[ test]
1279+ fn test_add_cross_file_link_collapses_one_line_asking_one_file_once ( ) {
1280+ let link = |target : & str , fragment : & str , line : usize | CrossFileLinkIndex {
1281+ target_path : target. to_string ( ) ,
1282+ fragment : fragment. to_string ( ) ,
1283+ line,
1284+ column : 1 ,
1285+ origin : LinkOrigin :: Body ,
1286+ } ;
1287+
1288+ let mut index = FileIndex :: new ( ) ;
1289+ index. add_cross_file_link ( link ( "target.md?raw=true" , "missing" , 3 ) ) ;
1290+ index. add_cross_file_link ( link ( "target.md?plain=1" , "missing" , 3 ) ) ;
1291+ assert_eq ! (
1292+ index. cross_file_links. len( ) ,
1293+ 1 ,
1294+ "one file, one fragment, one line is one entry, got: {:?}" ,
1295+ index. cross_file_links
1296+ ) ;
1297+
1298+ index. add_cross_file_link ( link ( "target.md" , "other" , 3 ) ) ;
1299+ index. add_cross_file_link ( link ( "target.md" , "missing" , 4 ) ) ;
1300+ assert_eq ! ( index. cross_file_links. len( ) , 3 ) ;
1301+ }
1302+
1303+ /// A link is a dependency on the file it names, so editing `b.md` re-lints
1304+ /// the source whichever way that source spelled the destination. The query
1305+ /// string is the case that gets this wrong: it is not part of a file name,
1306+ /// nothing ever creates a file called `b.md?raw=true`, and a reverse
1307+ /// dependency filed under that name is one no editor will ever look up.
1308+ #[ test]
1309+ fn test_reverse_deps_ignore_a_query_string_on_the_destination ( ) {
1310+ let mut index = WorkspaceIndex :: new ( ) ;
1311+
1312+ let mut file_a = FileIndex :: new ( ) ;
1313+ file_a. add_cross_file_link ( CrossFileLinkIndex {
1314+ target_path : "b.md?raw=true" . to_string ( ) ,
1315+ fragment : "section" . to_string ( ) ,
1316+ line : 10 ,
1317+ column : 5 ,
1318+ origin : LinkOrigin :: Body ,
1319+ } ) ;
1320+ index. update_file ( Path :: new ( "docs/a.md" ) , file_a) ;
1321+
1322+ assert_eq ! (
1323+ index. get_dependents( Path :: new( "docs/b.md" ) ) ,
1324+ vec![ PathBuf :: from( "docs/a.md" ) ] ,
1325+ "editing docs/b.md must re-lint the file linking to it"
1326+ ) ;
1327+
1328+ // And the source stops depending on it once the link is gone, so the
1329+ // stripped key is cleared by the same route it was created.
1330+ index. update_file ( Path :: new ( "docs/a.md" ) , FileIndex :: new ( ) ) ;
1331+ assert ! ( index. get_dependents( Path :: new( "docs/b.md" ) ) . is_empty( ) ) ;
1332+ }
1333+
11781334 #[ test]
11791335 fn test_reverse_deps_basic ( ) {
11801336 let mut index = WorkspaceIndex :: new ( ) ;
0 commit comments