diff --git a/DistFiles/localization/en/BloomMediumPriority.xlf b/DistFiles/localization/en/BloomMediumPriority.xlf index 011357681136..dc7ab59d9d79 100644 --- a/DistFiles/localization/en/BloomMediumPriority.xlf +++ b/DistFiles/localization/en/BloomMediumPriority.xlf @@ -1526,6 +1526,21 @@ ID: ImageLibrary.ThisPage Short link text used in setup instructions (e.g., as in "go to this page"). Should be lowercase as it appears mid-sentence. + + Bloom can no longer watch the Team Collection folder at "{0}", so it will not see changes made by your teammates. Usually this means the folder, or the drive or network it is on, is no longer available. + ID: TeamCollection.LostContactWithRepo + Shown in the Team Collection status dialog when Bloom loses its ability to watch the shared Team Collection folder part way through a session, for example because the network share or Dropbox folder disappears. {0} is replaced with the full path of the shared folder. "Bloom" is a product name and must not be translated. + + + Bloom can no longer see the Team Collection folder, so you will not see changes made by your teammates. + ID: TeamCollection.NoLongerSeeingChanges + A short pop-up notification (toast) in the bottom-right corner of the window, shown the moment Bloom notices part way through a session that it has lost contact with the shared Team Collection folder. Clicking it opens the Team Collection dialog, which shows the longer TeamCollection.LostContactWithRepo message. Keep it short; it has to fit in a small pop-up. "Bloom" is a product name and must not be translated. + + + Bloom may have missed some changes your teammates made. Please click "Reload Collection" to be sure you have the latest. + ID: TeamCollection.MayHaveMissedChanges + Used both as a pop-up notification (toast) and in the Team Collection status dialog, when so many files changed at once that Bloom's file-change notifications overflowed and some were lost. "Reload Collection" is the label of a button in that dialog and should match its translation there. "Bloom" is a product name and must not be translated. + Please wait while Bloom does some housekeeping on your book... ID: BookProcessor.HousekeepingMessage diff --git a/src/BloomBrowserUI/.prettierignore b/src/BloomBrowserUI/.prettierignore index 76da946114b8..3a5ac4e6ff6c 100644 --- a/src/BloomBrowserUI/.prettierignore +++ b/src/BloomBrowserUI/.prettierignore @@ -12,4 +12,8 @@ pnpm-lock.yaml # Overflow test fixtures, generated from their .pug (see the note there). Prettier would add # whitespace inside the test boxes, which changes what the overflow tests measure. -bookEdit/OverflowChecker/*Fixture.html +# The leading **/ matters: the pre-commit hook runs pretty-quick, which resolves the git +# root and passes repo-root-relative paths (src/BloomBrowserUI/bookEdit/...). A pattern +# anchored at this directory does not match those, so the fixtures got reformatted -- +# and four OverflowChecker tests failed -- on any commit that happened to stage them. +**/bookEdit/OverflowChecker/*Fixture.html diff --git a/src/BloomBrowserUI/collectionsTab/collectionsTabBookPane/CollectionsTabBookPane.tsx b/src/BloomBrowserUI/collectionsTab/collectionsTabBookPane/CollectionsTabBookPane.tsx index 728ac68ae30f..51dd07f3d30f 100644 --- a/src/BloomBrowserUI/collectionsTab/collectionsTabBookPane/CollectionsTabBookPane.tsx +++ b/src/BloomBrowserUI/collectionsTab/collectionsTabBookPane/CollectionsTabBookPane.tsx @@ -79,7 +79,9 @@ export const CollectionsTabBookPane: React.FunctionComponent<{ setBookTeamCollectionStatus((prevBookStatus) => ({ ...prevBookStatus, - disconnected: true, + // The field is isDisconnected; setting "disconnected" left the panel + // rendering the book as available for checkout. See BL-16729. + isDisconnected: true, error: errorMessage, })); }, diff --git a/src/BloomE2E/tests/reader-tool-stage-and-level.spec.ts b/src/BloomE2E/tests/reader-tool-stage-and-level.spec.ts index d37fed9616d2..c49fdca7ed35 100644 --- a/src/BloomE2E/tests/reader-tool-stage-and-level.spec.ts +++ b/src/BloomE2E/tests/reader-tool-stage-and-level.spec.ts @@ -45,113 +45,123 @@ test.beforeEach(async ({ page }) => { // The old toolbox code leans on that late restore to correct other things, so the fix is left to // the rework. toolbox-sections.spec.ts has the same problem and a fuller account. -test.fixme("each book remembers its own decodable stage [Test Case ID 442]", async ({ - page, -}) => { - const firstBook = await makeBasicBookWithReaderTool( - page, - "decodableReader", - ); - await setReaderPhase(page, "decodableReader", 2); - const secondBook = await makeBasicBookWithReaderTool( - page, - "decodableReader", - ); - await setReaderPhase(page, "decodableReader", 4); - - await editBook(page, firstBook); - await openReaderTool(page, "decodableReader"); - await expectReaderToolToShow( - page, - "decodableReader", - 2, - "The first book should come back on the stage it was left on.", - ); - - await editBook(page, secondBook); - await openReaderTool(page, "decodableReader"); - await expectReaderToolToShow( - page, - "decodableReader", - 4, - "The second book should come back on its own stage, not the first book's.", - ); -}); - -test.fixme("each book remembers its own level [Test Case ID 442]", async ({ - page, -}) => { - const firstBook = await makeBasicBookWithReaderTool(page, "leveledReader"); - await setReaderPhase(page, "leveledReader", 2); - const secondBook = await makeBasicBookWithReaderTool(page, "leveledReader"); - await setReaderPhase(page, "leveledReader", 3); - - await editBook(page, firstBook); - await openReaderTool(page, "leveledReader"); - await expectReaderToolToShow( - page, - "leveledReader", - 2, - "The first book should come back on the level it was left on.", - ); - - await editBook(page, secondBook); - await openReaderTool(page, "leveledReader"); - await expectReaderToolToShow( - page, - "leveledReader", - 3, - "The second book should come back on its own level, not the first book's.", - ); -}); - -test.fixme("a new book starts on the stage last chosen, not the one last seen [Test Case ID 441]", async ({ - page, -}) => { - const bookLeftOnStage2 = await makeBasicBookWithReaderTool( - page, - "decodableReader", - ); - await setReaderPhase(page, "decodableReader", 2); - await makeBasicBookWithReaderTool(page, "decodableReader"); - await setReaderPhase(page, "decodableReader", 4); - // Look at the first book again. Its stage 2 is restored, which must not become the default. - await editBook(page, bookLeftOnStage2); - await openReaderTool(page, "decodableReader"); - // sanity check: the restore we are guarding against really did show stage 2 - await expectReaderToolToShow( - page, - "decodableReader", - 2, - "The first book should have come back on stage 2 before the new book was made.", - ); - - await makeBookFromTemplate(page, "Decodable Reader"); - await openReaderTool(page, "decodableReader"); - - await expectReaderToolToShow( - page, - "decodableReader", - 4, - "A new decodable book should start on the stage last chosen with the arrows (4), " + - "not on the stage of the book last looked at (2), nor on stage 1.", - ); -}); - -test.fixme("a new book starts on the level last chosen [Test Case ID 460]", async ({ - page, -}) => { - await makeBasicBookWithReaderTool(page, "leveledReader"); - await setReaderPhase(page, "leveledReader", 3); - - // makeBasicBookWithReaderTool waits for the new book to show the level Bloom reports as its - // default, so what this test adds is that the default really is the level chosen above. - await makeBasicBookWithReaderTool(page, "leveledReader"); - - await expectReaderToolToShow( - page, - "leveledReader", - 3, - "A new book should start on the level last chosen, not on level 1.", - ); -}); +test.fixme( + "each book remembers its own decodable stage [Test Case ID 442]", + async ({ page }) => { + const firstBook = await makeBasicBookWithReaderTool( + page, + "decodableReader", + ); + await setReaderPhase(page, "decodableReader", 2); + const secondBook = await makeBasicBookWithReaderTool( + page, + "decodableReader", + ); + await setReaderPhase(page, "decodableReader", 4); + + await editBook(page, firstBook); + await openReaderTool(page, "decodableReader"); + await expectReaderToolToShow( + page, + "decodableReader", + 2, + "The first book should come back on the stage it was left on.", + ); + + await editBook(page, secondBook); + await openReaderTool(page, "decodableReader"); + await expectReaderToolToShow( + page, + "decodableReader", + 4, + "The second book should come back on its own stage, not the first book's.", + ); + }, +); + +test.fixme( + "each book remembers its own level [Test Case ID 442]", + async ({ page }) => { + const firstBook = await makeBasicBookWithReaderTool( + page, + "leveledReader", + ); + await setReaderPhase(page, "leveledReader", 2); + const secondBook = await makeBasicBookWithReaderTool( + page, + "leveledReader", + ); + await setReaderPhase(page, "leveledReader", 3); + + await editBook(page, firstBook); + await openReaderTool(page, "leveledReader"); + await expectReaderToolToShow( + page, + "leveledReader", + 2, + "The first book should come back on the level it was left on.", + ); + + await editBook(page, secondBook); + await openReaderTool(page, "leveledReader"); + await expectReaderToolToShow( + page, + "leveledReader", + 3, + "The second book should come back on its own level, not the first book's.", + ); + }, +); + +test.fixme( + "a new book starts on the stage last chosen, not the one last seen [Test Case ID 441]", + async ({ page }) => { + const bookLeftOnStage2 = await makeBasicBookWithReaderTool( + page, + "decodableReader", + ); + await setReaderPhase(page, "decodableReader", 2); + await makeBasicBookWithReaderTool(page, "decodableReader"); + await setReaderPhase(page, "decodableReader", 4); + // Look at the first book again. Its stage 2 is restored, which must not become the default. + await editBook(page, bookLeftOnStage2); + await openReaderTool(page, "decodableReader"); + // sanity check: the restore we are guarding against really did show stage 2 + await expectReaderToolToShow( + page, + "decodableReader", + 2, + "The first book should have come back on stage 2 before the new book was made.", + ); + + await makeBookFromTemplate(page, "Decodable Reader"); + await openReaderTool(page, "decodableReader"); + + await expectReaderToolToShow( + page, + "decodableReader", + 4, + "A new decodable book should start on the stage last chosen with the arrows (4), " + + "not on the stage of the book last looked at (2), nor on stage 1.", + ); + }, +); + +test.fixme( + "a new book starts on the level last chosen [Test Case ID 460]", + async ({ page }) => { + await makeBasicBookWithReaderTool(page, "leveledReader"); + await setReaderPhase(page, "leveledReader", 3); + + // makeBasicBookWithReaderTool waits for the new book to show the level Bloom reports as its + // default, so what this test adds is that the default really is the level chosen above. + await makeBasicBookWithReaderTool(page, "leveledReader"); + + await expectReaderToolToShow( + page, + "leveledReader", + 3, + "A new book should start on the level last chosen, not on level 1.", + ); + }, +); diff --git a/src/BloomE2E/tests/toolbox-sections.spec.ts b/src/BloomE2E/tests/toolbox-sections.spec.ts index a5d92ade8a4e..7f338ebbe70d 100644 --- a/src/BloomE2E/tests/toolbox-sections.spec.ts +++ b/src/BloomE2E/tests/toolbox-sections.spec.ts @@ -39,53 +39,56 @@ test.use({ // fixes the shutting half. The tool half is not fixed, and the old toolbox code leans on that late // restore to correct other things, so it is left to the rework. -test.fixme("turning a tool on under More... adds its section in order and opens it [Test Case ID 830]", async ({ - page, -}) => { - await makeBookFromTemplate(page, "Basic Book"); - await showToolbox(page); - // sanity check: a new Basic Book does not already have the Leveled Reader - expect( - await getShownTools(page), - "A new Basic Book should not start with the Leveled Reader tool.", - ).not.toContain("leveledReader"); +test.fixme( + "turning a tool on under More... adds its section in order and opens it [Test Case ID 830]", + async ({ page }) => { + await makeBookFromTemplate(page, "Basic Book"); + await showToolbox(page); + // sanity check: a new Basic Book does not already have the Leveled Reader + expect( + await getShownTools(page), + "A new Basic Book should not start with the Leveled Reader tool.", + ).not.toContain("leveledReader"); - await setToolTurnedOn(page, "leveledReader", true); + await setToolTurnedOn(page, "leveledReader", true); - // Bloom opens the tool just turned on, after a moment that lets the person see the box tick - // before "More..." closes (BL-16501). - await expectOpenTool( - page, - "leveledReader", - "Turning the Leveled Reader on did not open it.", - ); - // Sections are in alphabetical order of their labels, with "More..." always last. - const shown = await getShownTools(page); - expect( - shown.indexOf("leveledReader"), - `The Leveled Reader should come before Talking Book. The toolbox shows: ${shown.join(", ")}.`, - ).toBeLessThan(shown.indexOf("talkingBook")); - expect(shown[shown.length - 1], "More... should be the last section.").toBe( - "settings", - ); -}); + // Bloom opens the tool just turned on, after a moment that lets the person see the box tick + // before "More..." closes (BL-16501). + await expectOpenTool( + page, + "leveledReader", + "Turning the Leveled Reader on did not open it.", + ); + // Sections are in alphabetical order of their labels, with "More..." always last. + const shown = await getShownTools(page); + expect( + shown.indexOf("leveledReader"), + `The Leveled Reader should come before Talking Book. The toolbox shows: ${shown.join(", ")}.`, + ).toBeLessThan(shown.indexOf("talkingBook")); + expect( + shown[shown.length - 1], + "More... should be the last section.", + ).toBe("settings"); + }, +); -test.fixme("turning a tool off under More... removes its section [Test Case ID 830]", async ({ - page, -}) => { - const bookFolder = await makeBookFromTemplate(page, "Basic Book"); - await enableToolForBook(page, bookFolder, "leveledReader"); - // sanity check: the tool is there to be turned off - expect(await getShownTools(page)).toContain("leveledReader"); +test.fixme( + "turning a tool off under More... removes its section [Test Case ID 830]", + async ({ page }) => { + const bookFolder = await makeBookFromTemplate(page, "Basic Book"); + await enableToolForBook(page, bookFolder, "leveledReader"); + // sanity check: the tool is there to be turned off + expect(await getShownTools(page)).toContain("leveledReader"); - await setToolTurnedOn(page, "leveledReader", false); + await setToolTurnedOn(page, "leveledReader", false); - const shown = await getShownTools(page); - expect( - shown, - "Turning the Leveled Reader off should leave the other sections alone.", - ).toEqual(expect.arrayContaining(["talkingBook", "settings"])); -}); + const shown = await getShownTools(page); + expect( + shown, + "Turning the Leveled Reader off should leave the other sections alone.", + ).toEqual(expect.arrayContaining(["talkingBook", "settings"])); + }, +); test("clicking the header of the open tool leaves it open [Test Case ID 830]", async ({ page, @@ -100,34 +103,35 @@ test("clicking the header of the open tool leaves it open [Test Case ID 830]", a expect(await getOpenTool(page)).toBe(open); }); -test.fixme("the book remembers which tool was open [Test Case ID 830]", async ({ - page, -}) => { - const bookFolder = await makeBookFromTemplate(page, "Basic Book"); - await enableToolForBook(page, bookFolder, "leveledReader"); - await showToolbox(page); - // sanity check: the tool we will look for is not the one the book opens anyway - expect( - await waitForOpenTool(page), - "The book should not open the Leveled Reader before it has been opened once.", - ).not.toBe("leveledReader"); - await openReaderTool(page, "leveledReader"); - expect(await getOpenTool(page)).toBe("leveledReader"); +test.fixme( + "the book remembers which tool was open [Test Case ID 830]", + async ({ page }) => { + const bookFolder = await makeBookFromTemplate(page, "Basic Book"); + await enableToolForBook(page, bookFolder, "leveledReader"); + await showToolbox(page); + // sanity check: the tool we will look for is not the one the book opens anyway + expect( + await waitForOpenTool(page), + "The book should not open the Leveled Reader before it has been opened once.", + ).not.toBe("leveledReader"); + await openReaderTool(page, "leveledReader"); + expect(await getOpenTool(page)).toBe("leveledReader"); - await editBook(page, bookFolder); - // The book keeps it in its meta.json, not just in memory, so it survives a restart of Bloom. - const meta = JSON.parse( - fs.readFileSync(Path.join(bookFolder, "meta.json"), "utf8"), - ) as { currentTool?: string }; - expect( - meta.currentTool, - "The book's meta.json should record the Leveled Reader as its open tool.", - ).toMatch(/^leveledReader/); - await showToolbox(page); + await editBook(page, bookFolder); + // The book keeps it in its meta.json, not just in memory, so it survives a restart of Bloom. + const meta = JSON.parse( + fs.readFileSync(Path.join(bookFolder, "meta.json"), "utf8"), + ) as { currentTool?: string }; + expect( + meta.currentTool, + "The book's meta.json should record the Leveled Reader as its open tool.", + ).toMatch(/^leveledReader/); + await showToolbox(page); - await expectOpenTool( - page, - "leveledReader", - "Coming back to the book should reopen the tool that was open when it was left.", - ); -}); + await expectOpenTool( + page, + "leveledReader", + "Coming back to the book should reopen the tool that was open when it was left.", + ); + }, +); diff --git a/src/BloomExe/Program.cs b/src/BloomExe/Program.cs index 591ba656bdde..9b6e1a4e9391 100644 --- a/src/BloomExe/Program.cs +++ b/src/BloomExe/Program.cs @@ -2910,6 +2910,46 @@ private static bool IsBenignSocketAbortExceptionType(string exceptionTypeName) } } + /// + /// Run the action on the UI thread on a LATER turn of the message pump. + /// + /// It must be later, not inline, because one caller is TryStartWatching in the middle of + /// StartMonitoring: disconnecting inline there would tear the watchers down while that + /// method is still setting them up. And it must not block, because the other callers are + /// on a file system watcher's thread or the heartbeat's, where a synchronous invoke can + /// deadlock against a UI thread waiting on the BloomServer. + /// + /// Program.MainContext is the WinForms synchronization context, captured once at startup. + /// Post satisfies both requirements, is safe to call from any thread, and -- unlike + /// looking a form up -- never enumerates Application.OpenForms, which is not thread-safe + /// and which every caller here would be enumerating from a background thread. + /// + internal static void RunOnUiThreadLater(Action action) + { + var uiContext = MainContext; + if (uiContext == null) + { + // No UI thread exists: unit tests, or startup before Application.Run. Nobody can + // be racing us, and the state change matters more than the notification, so just + // do it here. + action(); + return; + } + try + { + uiContext.Post(_ => action(), null); + } + catch (Exception ex) + { + // Each action() is responsible for its own exception handling as needed. + // This catch is largely a safety net for the Post itself, handling the case where + // the uiContext has been disposed just as the Post is being called. We don't want + // to crash the app because of that, and we do want to report it to Sentry because + // it is unexpected and should be investigated. + NonFatalProblem.ReportSentryOnly(ex); + } + } + /// ------------------------------------------------------------------------------------ internal static void SetUpErrorHandling() { diff --git a/src/BloomExe/RobustFileIO.cs b/src/BloomExe/RobustFileIO.cs index f3e57be613d2..e775ecdad7c1 100644 --- a/src/BloomExe/RobustFileIO.cs +++ b/src/BloomExe/RobustFileIO.cs @@ -1,4 +1,5 @@ using System; +using System.Collections.Generic; using System.IO; using Bloom.ImageProcessing; using SIL.Code; @@ -82,6 +83,38 @@ public static void SaveTaglibFile(TagLib.File file) /// Attempts to delete a directory once. Returns false if an exception prevents deletion, /// rather than retrying or throwing — suitable for best-effort cleanup during shutdown. /// + /// + /// Append to a file, retrying only briefly. RobustFile.AppendAllText keeps trying for + /// seconds, which is right when nothing is waiting on the answer and wrong when the UI + /// thread is: Bloom's Team Collection history is written from Application.Idle, so a log + /// file that something else has open (a backup agent, an anti-virus scan) would freeze + /// the window for as long as RobustFile persisted. This makes the same underlying call + /// -- and so writes the same encoding, UTF-8 with no BOM, letting the two interleave in + /// one file -- but gives up after a couple of quick tries and lets the caller decide what + /// to do with what it could not write. See BL-16729. + /// + /// It lives here rather than at the call site because this file is where Bloom keeps its + /// deliberate, reviewed uses of the raw File API. + /// + /// Typically just IOException: a permissions failure + /// will not improve while we wait, so it should come straight back out. + public static void QuickAppendAllText( + string path, + string text, + int attempts, + int retryDelayMs, + ISet exceptionTypesToRetry + ) + { + RetryUtility.Retry( + () => File.AppendAllText(path, text), + attempts, + retryDelayMs, + exceptionTypesToRetry, + memo: $"AppendAllText {path}" + ); + } + public static bool TryDeleteDirectory(string path, bool recursive = false) { try diff --git a/src/BloomExe/TeamCollection/ConnectionFailureTracker.cs b/src/BloomExe/TeamCollection/ConnectionFailureTracker.cs new file mode 100644 index 000000000000..d210b772b7a5 --- /dev/null +++ b/src/BloomExe/TeamCollection/ConnectionFailureTracker.cs @@ -0,0 +1,56 @@ +namespace Bloom.TeamCollection +{ + /// + /// Decides when a run of failed connection checks has gone on long enough to believe. + /// Kept as a separate, pure class (no IO, no timers, no threads) so the policy can be + /// unit tested exhaustively; ConnectionHeartbeat supplies the timing. See BL-16729. + /// + /// The point of waiting for a second failure is that the things CheckConnection looks at + /// can lie: one dropped packet fails the probe to dropbox.com, a Wi-Fi roam briefly makes + /// NetworkInterface.GetIsNetworkAvailable() false, a flaky SMB share can answer "no such + /// folder" for a moment. Wrongly disconnecting a collection that is working is the worst + /// outcome available to us, because there is no automatic way back: the user has to + /// Reload Collection. Waiting fifteen seconds to be sure is cheap by comparison. + /// + internal class ConnectionFailureTracker + { + /// + /// How many checks in a row must report the same problem before we act on it. + /// + internal const int kRequiredConsecutiveFailures = 2; + + private string _lastFailureL10nId; + private int _consecutiveFailures; + + /// + /// Feed in the result of one connection check. Returns true when we have now seen + /// enough consecutive failures of the same kind to conclude we really are disconnected. + /// + /// What CheckConnection returned: null means all is well. + public bool RecordResult(TeamCollectionMessage problemOrNull) + { + if (problemOrNull == null) + { + Reset(); + return false; + } + if (problemOrNull.L10NId != _lastFailureL10nId) + { + // A different problem from last time. "No network" followed by "repo missing" + // is two transients, not one sustained outage, so start counting again. + _lastFailureL10nId = problemOrNull.L10NId; + _consecutiveFailures = 0; + } + return ++_consecutiveFailures >= kRequiredConsecutiveFailures; + } + + /// + /// Forget any run of failures, e.g. because we stopped checking for a while. + /// + public void Reset() + { + _consecutiveFailures = 0; + _lastFailureL10nId = null; + } + } +} diff --git a/src/BloomExe/TeamCollection/ConnectionHeartbeat.cs b/src/BloomExe/TeamCollection/ConnectionHeartbeat.cs new file mode 100644 index 000000000000..ee49928befd8 --- /dev/null +++ b/src/BloomExe/TeamCollection/ConnectionHeartbeat.cs @@ -0,0 +1,173 @@ +using System; +using System.Threading; + +namespace Bloom.TeamCollection +{ + /// + /// Periodically re-checks that we can still reach the Team Collection repo, and disconnects + /// if we can't. See BL-16729. + /// + /// The file system watchers tell us at once when the shared folder is yanked away, but they + /// cannot tell us that Dropbox has stopped syncing: the folder is still there, we simply + /// stop receiving other people's work. Before this, CheckConnection ran only when the user + /// did something (check out, check in, delete), so somebody who was just reading and editing + /// their own checked-out book could go the whole session without finding out. + /// + /// Owned by TeamCollection.StartMonitoring/StopMonitoring, which gets the lifecycle right + /// for free: no heartbeat during SyncAtStartup (monitoring is deliberately off then), none + /// on a DisconnectedTeamCollection (whose Start/StopMonitoring are no-ops), and it stops + /// when we disconnect or dispose. + /// + internal sealed class ConnectionHeartbeat : IDisposable + { + /// + /// How long between checks when everything is fine. Long enough that the cost (which for + /// a Dropbox repo includes an HTTP HEAD to dropbox.com) is negligible, short enough that + /// the user finds out reasonably soon that they have stopped seeing their teammates' work. + /// Not const, and internal, so tests can shorten it. + /// + internal static int IntervalMs = 60 * 1000; + + /// + /// How soon we look again after a check fails, to see whether it was just a blip. + /// Must be longer than DropboxUtils' 10-second cache of the dropbox.com probe, or the + /// second look would just return the first one's answer and confirm nothing. + /// + internal const int kConfirmIntervalMs = 15 * 1000; + + private readonly TeamCollection _teamCollection; + private readonly ConnectionFailureTracker _tracker = new ConnectionFailureTracker(); + private Timer _timer; + private volatile bool _disposed; + + /// + /// Whether Start() actually armed a timer. Exists so the test for the RunningUnitTests + /// guard can see the guard work: the first tick would be a minute away, so counting + /// probes just after Start() would read zero whether the guard was there or not. + /// + internal bool IsTimerArmed => _timer != null; + + public ConnectionHeartbeat(TeamCollection teamCollection) + { + _teamCollection = teamCollection; + } + + /// + /// Begin checking. Does nothing under unit tests, which must not be left with live + /// threadpool timers checking real folders and the real network. + /// + internal void Start() + { + if (Program.RunningUnitTests) + return; + // A one-shot timer that re-arms itself at the end of each tick (rather than a + // repeating one) makes overlapping ticks structurally impossible, so a probe that + // blocks for forty seconds on a dead share cannot pile up behind itself. + _timer = new Timer(UpdateTcConnectionStatus, null, IntervalMs, Timeout.Infinite); + } + + /// + /// Runs on a threadpool thread. Internal so tests can drive the policy directly, with + /// no timer involved. + /// + internal void UpdateTcConnectionStatus(object unused) + { + var delayUntilNextTick = IntervalMs; + try + { + if (!OkToCheckNow()) + { + // Not the live collection, not watching just now (e.g. during a sync), or + // busy writing to the repo. Anything we noticed before such a gap is no + // longer part of a "consecutive" run, and an in-flight write reports its + // own failures, so disconnecting out from under it would be disruptive. + _tracker.Reset(); + } + else + { + // A joiner's Books folder can arrive minutes after Bloom starts, once + // Dropbox delivers it. If we could not watch it then, this is where we + // notice it has turned up. See BL-16729. + _teamCollection.RetryDeferredWatching(); + + // Deliberately the quiet overload: History messages are not de-duplicated, + // so a probe that wrote them would fill log.txt and raise a status-changed + // event on every tick of a perfectly healthy session. + var problem = _teamCollection.CheckConnection(writeHistoryMessages: false); + // The probe can block for several seconds (DropboxUtils allows 5 for its + // request to dropbox.com), which is long enough for a check-in or a sync to + // have started meanwhile. Re-ask before acting on what we found, or we + // would disconnect in the middle of one. + if (!OkToCheckNow()) + { + _tracker.Reset(); + } + else if (_tracker.RecordResult(problem)) + { + // Logged as well as acted on: without this, someone testing a real + // outage has no way to tell "the check ran and decided" from "the + // check never ran". + SIL.Reporting.Logger.WriteEvent( + $"Team Collection periodic check confirmed a problem ({problem.L10NId}); disconnecting." + ); + _teamCollection.ReportConnectionProblem(problem); + } + else if (problem != null) + { + SIL.Reporting.Logger.WriteEvent( + $"Team Collection periodic check found a problem ({problem.L10NId}); " + + $"looking again in {kConfirmIntervalMs / 1000}s before believing it." + ); + delayUntilNextTick = kConfirmIntervalMs; // suspicious; confirm sooner + } + } + } + catch (Exception ex) + { + // An exception here is not evidence that the repo is gone, so we don't + // disconnect over it. This matches TeamCollectionManager.CheckConnection. + NonFatalProblem.ReportSentryOnly(ex); + // It is also not evidence that the repo is FINE, so this tick tells us nothing + // either way -- which means it breaks the run. Without this reset, a failure, + // then a throwing probe, then another failure would count as two consecutive + // failures and disconnect a collection that was never shown to be unreachable + // twice in a row. + _tracker.Reset(); + } + finally + { + if (!_disposed) + { + try + { + _timer?.Change(delayUntilNextTick, Timeout.Infinite); + } + catch (ObjectDisposedException) + { + // Disposed while we were checking. Nothing to re-arm. + } + } + } + } + + /// + /// Whether this is a sensible moment to check the connection at all. Checked both + /// before and after the probe, because the probe can block long enough for the answer + /// to change. + /// + private bool OkToCheckNow() + { + return !_disposed + && _teamCollection.IsMonitoring + && _teamCollection.IsLiveCollection + && !_teamCollection.IsWritingToRepo; + } + + public void Dispose() + { + _disposed = true; + _timer?.Dispose(); + _timer = null; + } + } +} diff --git a/src/BloomExe/TeamCollection/FolderTeamCollection.cs b/src/BloomExe/TeamCollection/FolderTeamCollection.cs index 4f9b1bfee045..e62e45656e27 100644 --- a/src/BloomExe/TeamCollection/FolderTeamCollection.cs +++ b/src/BloomExe/TeamCollection/FolderTeamCollection.cs @@ -1,4 +1,4 @@ -using System; +using System; using System.Collections.Generic; using System.Diagnostics; using System.IO; @@ -144,37 +144,51 @@ protected override void PutBookInRepo( _writeBookInProgress = true; } + // The finally is what guarantees _writeBookInProgress gets cleared even when the zip + // or the Replace throws. Without it a single failed write left the flag set for the + // rest of the session, which permanently suppressed change notifications for this + // book and (since BL-16729) would have silently killed the periodic connection + // check -- in exactly the flaky-share situation that check exists to catch. try { - var zipFile = new BloomZipFile(pathToWrite); - zipFile.AddDirectory( - sourceBookFolderPath, - sourceBookFolderPath.Length + 1, - null, - progressCallback - ); - zipFile.SetComment(status.WithCollectionId(CollectionId).ToJson()); - zipFile.Save(); - // If by any chance we've previously created a tombstone for this book, get rid of it. - var pathForTombstone = GetPathForTombstone(bookFolderName); - if (pathForTombstone != null) - RobustFile.Delete(pathForTombstone); - } - catch (Exception) - { - RobustFile.Delete(pathToWrite); // try to clean up - throw; - } + try + { + var zipFile = new BloomZipFile(pathToWrite); + zipFile.AddDirectory( + sourceBookFolderPath, + sourceBookFolderPath.Length + 1, + null, + progressCallback + ); + zipFile.SetComment(status.WithCollectionId(CollectionId).ToJson()); + zipFile.Save(); + // If by any chance we've previously created a tombstone for this book, get rid of it. + var pathForTombstone = GetPathForTombstone(bookFolderName); + if (pathForTombstone != null) + RobustFile.Delete(pathForTombstone); + } + catch (Exception) + { + RobustFile.Delete(pathToWrite); // try to clean up + throw; + } - if (pathToWrite != bookPath) - { - RobustFile.Replace(pathToWrite, bookPath, null); - } + if (pathToWrite != bookPath) + { + RobustFile.Replace(pathToWrite, bookPath, null); + } - lock (_lockObject) + lock (_lockObject) + { + _lastWriteBookTime = DateTime.Now; + } + } + finally { - _lastWriteBookTime = DateTime.Now; - _writeBookInProgress = false; + lock (_lockObject) + { + _writeBookInProgress = false; + } } } @@ -647,6 +661,20 @@ public override bool DoLocalAndRemoteNamesDifferOnlyByCase(string bookBaseName) public override string RepoDescription => _repoFolderPath; + /// + /// As well as a sync, a book write in progress counts as "busy with the repo". + /// + protected internal override bool IsWritingToRepo + { + get + { + lock (_lockObject) + { + return base.IsWritingToRepo || _writeBookInProgress; + } + } + } + // The standard place where we store zip files for a collection-level folder. private static string GetZipFileForFolder(string folderName, string repoFolderPath) { @@ -976,11 +1004,90 @@ static void ExtractFolder(string collectionFolder, string repoFolder, string fol protected internal override void StartMonitoring() { base.StartMonitoring(); - _booksWatcher = new FileSystemWatcherWrapper(); var booksPath = Path.Combine(_repoFolderPath, "Books"); - if (!Directory.Exists(booksPath)) - return; // probably joining a TC and didn't get it synced properly. + if (Directory.Exists(booksPath)) + { + bool startedWatching; + try + { + startedWatching = StartBooksWatcher(booksPath); + } + catch (Exception ex) + { + // The folder can go between the test just above and FileSystemWatcher.Path + // taking hold of it, and assigning Path to a folder that is not there throws. + // Left alone that escapes StartMonitoring, so nobody is told and we do not go + // Disconnected. Report it exactly as a watcher that refuses to start is + // reported -- which is what RetryDeferredWatching already does for the same + // race on its own path. See BL-16729. + HandleRepoWatcherError(booksPath, ex); + return; + } + if (!startedWatching) + { + // StartBooksWatcher reported the failure, which (when there is no window to + // marshal to, e.g. at startup) disconnects us synchronously and calls + // StopMonitoring under us. Carrying on to create the Other watcher would + // leave a live watcher on a collection we have already given up on, which + // StopMonitoring has finished with and Dispose will not revisit. + return; + } + } + else + { + // BL-16729. Note this is a narrow case. The Books folder is created when the + // collection is set up, and you cannot *join* a collection without it -- the + // join fails earlier, in GetBookList, which throws on the missing folder + // (confirmed by testing, 2026-09-10). What remains reachable is opening or + // reloading a collection you have already joined at a moment when Books is + // absent, e.g. while Dropbox is still restoring the folder onto a new machine. + // SyncAtStartup will have failed and told the user; what this avoids is needing + // to restart Bloom once the folder finally arrives. + // + // Previously we gave up on watching for books for the rest of the session, and + // as a side effect skipped the Other watcher below as well. Now only the books + // watcher is deferred, and the periodic connection check retries it via + // RetryDeferredWatching. + // + // We deliberately do NOT treat this as a disconnection; that was considered and + // rejected, because no teammate can have checked a book in while there is + // nowhere for a book to be. + _booksWatcherDeferred = true; + Logger.WriteEvent( + $"Team Collection: \"{booksPath}\" does not exist yet, so book changes cannot be watched. Will retry." + ); + } + + var otherFilesDirPath = Path.Combine(_repoFolderPath, "Other"); + // If it doesn't exist we can't watch it. Rather bizarre since we normally create + // it if it doesn't exist as part of syncing. But BL-15838 seems to have been + // caused by not checking. If we can't set it up, unfortunately we won't find + // out immediately if some remote user modifies something in the collection. + // But we should find out on the next startup, and from then on we'll be able to + // monitor it, so I don't think it's very serious. + if (Directory.Exists(otherFilesDirPath)) + { + _otherWatcher = new FileSystemWatcherWrapper(otherFilesDirPath); + _otherWatcher.NotifyFilter = NotifyFilters.LastWrite; + _otherWatcher.InternalBufferSize = kWatcherBufferSize; + _otherWatcher.DebounceChanged(OnCollectionFilesChanged, kDebouncePeriodInMs); + _otherWatcher.Error += (sender, args) => + HandleRepoWatcherError(otherFilesDirPath, args.GetException()); + TryStartWatching(_otherWatcher, otherFilesDirPath); + } + } + + // True when StartMonitoring found no Books folder to watch. See BL-16729. + private bool _booksWatcherDeferred; + + /// + /// Set up and start the watcher on the repo's Books folder. Returns false if we could + /// not start watching (which will already have been reported). + /// + private bool StartBooksWatcher(string booksPath) + { + _booksWatcher = new FileSystemWatcherWrapper(); _booksWatcher.Path = booksPath; // Enhance: maybe one day we want to watch collection files too? @@ -989,28 +1096,201 @@ protected internal override void StartMonitoring() // the renaming of files or directories. _booksWatcher.NotifyFilter = NotifyFilters.LastWrite | NotifyFilters.FileName | NotifyFilters.DirectoryName; + // A Dropbox sync can land a whole batch of books at once, which is just the sort of + // burst that overflows the default 8KB buffer and silently loses notifications. + _booksWatcher.InternalBufferSize = kWatcherBufferSize; _booksWatcher.DebounceChanged(OnChanged, kDebouncePeriodInMs); _booksWatcher.DebounceCreated(OnCreated, kDebouncePeriodInMs); _booksWatcher.DebounceRenamed(OnRenamed, kDebouncePeriodInMs); _booksWatcher.DebounceDeleted(OnDeleted, kDebouncePeriodInMs); + // BL-16729: without this, a share that goes away mid-session kills the watch and + // nobody ever finds out. + _booksWatcher.Error += (sender, args) => + HandleRepoWatcherError(booksPath, args.GetException()); - // Begin watching. - _booksWatcher.EnableRaisingEvents = true; + return TryStartWatching(_booksWatcher, booksPath); + } - var otherFilesDirPath = Path.Combine(_repoFolderPath, "Other"); - // If it doesn't exist we can't watch it. Rather bizarre since we normally create - // it if it doesn't exist as part of syncing. But BL-15838 seems to have been - // caused by not checking. If we can't set it up, unfortunately we won't find - // out immediately if some remote user modifies something in the collection. - // But we should find out on the next startup, and from then on we'll be able to - // monitor it, so I don't think it's very serious. - if (Directory.Exists(otherFilesDirPath)) + /// + /// The Books folder was not there when we started monitoring, so we have no watcher for + /// book changes. Called from the periodic connection check: if Dropbox has delivered the + /// folder since, start watching it now. See BL-16729. + /// + protected internal override void RetryDeferredWatching() + { + // Cheap tests first: this runs on the heartbeat's thread every minute, and for + // almost every collection there is nothing deferred. + if (!_booksWatcherDeferred || !IsMonitoring) + return; + var booksPath = Path.Combine(_repoFolderPath, "Books"); + if (!Directory.Exists(booksPath)) + return; + + // Starting the watcher and announcing the books we find raises the same events the + // watcher itself raises, which reach the UI, so hand the work to the UI thread + // rather than doing it on the heartbeat's thread-pool thread. + Program.RunOnUiThreadLater(() => { - _otherWatcher = new FileSystemWatcherWrapper(otherFilesDirPath); - _otherWatcher.NotifyFilter = NotifyFilters.LastWrite; - _otherWatcher.DebounceChanged(OnCollectionFilesChanged, kDebouncePeriodInMs); - _otherWatcher.EnableRaisingEvents = true; + // Re-check: we may have been stopped or disconnected while this was queued. + if (!_booksWatcherDeferred || !IsMonitoring || !IsLiveCollection) + return; + // Dropbox can take the folder away again between the test above, which ran on + // the heartbeat's thread, and this, which runs later on the UI thread. Giving a + // watcher a Path that is not there throws, and nothing catches an exception out + // of a posted callback, so that would take Bloom down. Stay deferred instead and + // let the next tick try again. + if (!Directory.Exists(booksPath)) + return; + _booksWatcherDeferred = false; + bool startedWatching; + try + { + startedWatching = StartBooksWatcher(booksPath); + } + catch (Exception ex) + { + // The same race, in the much narrower window between the test just above and + // the watcher taking hold of the folder. + _booksWatcherDeferred = true; + NonFatalProblem.ReportSentryOnly(ex); + return; + } + if (!startedWatching) + { + // The watcher threw an exception instead of starting, which is already reported to + // the user, changing the state of the program and display. It's too tricky to try + // again later, so give up on watching for the rest of this session. + return; + } + // Monitoring can have stopped while we were getting here: StopMonitoring runs on + // the sync progress dialog's worker thread (see SynchronizeRepoAndLocal), not the + // UI thread we are on. Its watcher teardown has then already happened and will not + // happen again -- Dispose only calls StopMonitoring while _monitoring is true -- so + // the watcher we have just enabled would go on raising events into a collection + // nobody is using. Take it down ourselves. Reading the field into a local keeps + // this safe if StopMonitoring gets to it first. + if (!IsMonitoring) + { + var strandedWatcher = _booksWatcher; + _booksWatcher = null; + if (strandedWatcher != null) + { + strandedWatcher.EnableRaisingEvents = false; + strandedWatcher.Dispose(); + } + return; + } + Logger.WriteEvent( + $"Team Collection: \"{booksPath}\" has appeared; now watching it for book changes." + ); + NoticeChangesToTheRepoBeforeWeStartedWatching(); + }); + } + + /// + /// We have just started watching the Books folder, having not been watching it since + /// Bloom started. No Created or Deleted event was raised for anything that happened in + /// the meantime, so work out what changed and raise those events now, in both + /// directions: books that arrived, and books that went away. + /// + private void NoticeChangesToTheRepoBeforeWeStartedWatching() + { + string[] bookNames; + try + { + bookNames = GetBookList(); + } + catch (Exception ex) + { + // The folder existed a moment ago; if it has gone again there is nothing to + // announce, and the watcher's own Error handling covers a real loss of contact. + NonFatalProblem.ReportSentryOnly(ex); + return; + } + NoticeBooksThatArrivedBeforeWeStartedWatching(bookNames); + NoticeBooksThatLeftTheRepoBeforeWeStartedWatching(bookNames); + } + + /// + /// Whatever is in the Books folder now got there while we had no watcher on it, so no + /// Created event was ever raised for any of it. From the point of view of watching that + /// folder these books are all new since Bloom started, so tell the rest of Bloom about + /// them exactly as the watcher would have. + /// + private void NoticeBooksThatArrivedBeforeWeStartedWatching(string[] bookNames) + { + foreach (var bookName in bookNames) + { + if (!Directory.Exists(Path.Combine(_localCollectionFolder, bookName))) + RaiseNewBook(bookName + ".bloom"); + else if (HasBeenChangedRemotely(bookName) || HasCheckoutChangedRemotely(bookName)) + HandleModifiedFile( + new BookRepoChangeEventArgs { BookFileName = bookName + ".bloom" } + ); + } + } + + /// + /// The other half of NoticeChangesToTheRepoBeforeWeStartedWatching: a book that was + /// deleted from the repo while we had no watcher raised no Deleted event either, so + /// nothing has told us the local copy is obsolete. Raise the event for each local book + /// whose repo counterpart is missing, exactly as OnDeleted would have. + /// + /// + /// We deliberately do not decide here whether the book really was deleted. A Books + /// folder that Dropbox is still restoring can be missing books that are perfectly + /// alive, so "absent from the repo" is not evidence of anything. HandleDeletedRepoFile + /// is where that judgment already lives: it waits, re-checks the repo, and only + /// recycles the local copy if it finds a tombstone saying someone deliberately deleted + /// the book. That also makes the event harmless for a book that was renamed remotely + /// (a rename leaves no tombstone), whose new name the loop above has already announced + /// as a new book, which is how a remote rename is normally detected. + /// + private void NoticeBooksThatLeftTheRepoBeforeWeStartedWatching(string[] repoBookNames) + { + // The Directory.Exists() test above is case-insensitive on Windows, and a book + // whose local name differs from the repo's only by case is not a deletion, so + // match names the same way here. + var namesInRepo = new HashSet(repoBookNames, StringComparer.OrdinalIgnoreCase); + string[] localFolders; + try + { + localFolders = Directory.GetDirectories(_localCollectionFolder); + } + catch (Exception ex) + { + NonFatalProblem.ReportSentryOnly(ex); + return; + } + foreach (var folderPath in localFolders) + { + var bookName = Path.GetFileName(folderPath); + if (namesInRepo.Contains(bookName)) + continue; + try + { + if (!IsBloomBookFolder(folderPath)) + continue; + // A book created here and never checked in has no repo counterpart to + // lose. Deliberately, such a book has no local status file until it is + // checked in; that is exactly how we tell it from one deleted remotely. + if (!IsBookKnownToTeamCollection(folderPath)) + continue; + // A book renamed here but not yet checked in is still in the repo under + // its old name, so it has not gone anywhere. + var oldName = GetLocalStatus(bookName).oldName; + if (!string.IsNullOrEmpty(oldName) && namesInRepo.Contains(oldName)) + continue; + } + catch (Exception ex) + { + // A book we cannot read is a book we have no business deleting; leave it + // for the next reload to sort out. + NonFatalProblem.ReportSentryOnly(ex); + continue; + } + RaiseDeleteRepoBookFile(bookName + ".bloom"); } } @@ -1197,6 +1477,9 @@ private void OnRenamed(object sender, FileSystemEventArgs e) protected internal override void StopMonitoring() { + // Whatever we were waiting to start watching, we are not waiting any more. + _booksWatcherDeferred = false; + if (_booksWatcher != null) { _booksWatcher.EnableRaisingEvents = false; @@ -1319,13 +1602,21 @@ protected override void WriteBookStatusJsonToRepo(string bookName, string status _lastWriteBookTime = DateTime.Now; } - // We've had some failures on very fast clicking of Checkin/Checkout. - // Not clear how they come to overlap, but it's worth just trying again - // as a recovery strategy. - RobustZip.WriteZipComment(status, bookPath); - lock (_lockObject) + try { - _writeBookInProgress = false; + // We've had some failures on very fast clicking of Checkin/Checkout. + // Not clear how they come to overlap, but it's worth just trying again + // as a recovery strategy. + RobustZip.WriteZipComment(status, bookPath); + } + finally + { + // See the note on the same pattern in PutBookInRepo: a throw here used to leave + // the flag set for the rest of the session. + lock (_lockObject) + { + _writeBookInProgress = false; + } } } @@ -1616,7 +1907,7 @@ string teamCollectionFolder /// /// Returns null if connection is fine, otherwise, a message describing the problem. /// - public override TeamCollectionMessage CheckConnection() + public override TeamCollectionMessage CheckConnection(bool writeHistoryMessages) { if (!Directory.Exists(_repoFolderPath)) { @@ -1643,11 +1934,14 @@ public override TeamCollectionMessage CheckConnection() if (!DropboxUtils.IsDropboxProcessRunning()) { if (isOnLocalNetwork) - _tcManager.MessageLog.WriteMessage( - MessageAndMilestoneType.History, - "TeamCollection.NeedDropboxRunningButLANOK", - "Dropbox does not appear to be running, but the folder has also been shared locally which appears to be okay." - ); + { + if (writeHistoryMessages) + _tcManager.MessageLog.WriteMessage( + MessageAndMilestoneType.History, + "TeamCollection.NeedDropboxRunningButLANOK", + "Dropbox does not appear to be running, but the folder has also been shared locally which appears to be okay." + ); + } else return new TeamCollectionMessage( MessageAndMilestoneType.Error, @@ -1659,11 +1953,14 @@ public override TeamCollectionMessage CheckConnection() if (!DropboxUtils.CanAccessDropbox()) { if (isOnLocalNetwork) - _tcManager.MessageLog.WriteMessage( - MessageAndMilestoneType.History, - "TeamCollection.NeedDropboxAccessButLANOK", - "Bloom cannot reach Dropbox.com, but the folder has also been shared locally which appears to be okay." - ); + { + if (writeHistoryMessages) + _tcManager.MessageLog.WriteMessage( + MessageAndMilestoneType.History, + "TeamCollection.NeedDropboxAccessButLANOK", + "Bloom cannot reach Dropbox.com, but the folder has also been shared locally which appears to be okay." + ); + } else return new TeamCollectionMessage( MessageAndMilestoneType.Error, diff --git a/src/BloomExe/TeamCollection/TeamCollection.cs b/src/BloomExe/TeamCollection/TeamCollection.cs index 94d30a7f857a..649cab0b449f 100644 --- a/src/BloomExe/TeamCollection/TeamCollection.cs +++ b/src/BloomExe/TeamCollection/TeamCollection.cs @@ -153,11 +153,23 @@ protected abstract void PutBookInRepo( public abstract bool KnownToHaveBeenDeleted(string oldName); + /// + /// Returns null if connection to repo is fine, otherwise, a message describing the problem. + /// + public TeamCollectionMessage CheckConnection() + { + return CheckConnection(true); + } + /// /// Returns null if connection to repo is fine, otherwise, a message describing the problem. /// This default implementation assumes nothing useful can be done to check the connection. /// - public virtual TeamCollectionMessage CheckConnection() + /// Pass false for a side-effect-free probe. The + /// periodic connection check (see ConnectionHeartbeat) calls this many times over a + /// session; History messages are not de-duplicated, so a probe that wrote them would + /// fill up log.txt and raise a status-changed event on every tick. + public virtual TeamCollectionMessage CheckConnection(bool writeHistoryMessages) { return null; } @@ -449,6 +461,48 @@ public string CopyBookFromRepoToLocal( private bool _monitoring = false; + // Periodically re-checks that the repo is still reachable, for the cases the file system + // watchers cannot detect (notably Dropbox stopping). Only non-null while monitoring. + private ConnectionHeartbeat _heartbeat; + + /// + /// True between StartMonitoring and StopMonitoring. Note that monitoring is deliberately + /// off during SyncAtStartup, so this is not the same as "this is the live collection". + /// + protected internal bool IsMonitoring => _monitoring; + + /// + /// True while we are in the middle of writing to the repo. The periodic connection check + /// skips its tick while this is true: an in-flight write reports its own failures, and + /// disconnecting out from under it would be both redundant and disruptive. + /// + protected internal virtual bool IsWritingToRepo => _syncIsRunning; + + /// + /// True if this is the collection the manager is currently using. False for one we have + /// been disconnected from but which has not been disposed yet. Virtual so tests can + /// drive the periodic connection check without a whole live TeamCollectionManager. + /// + protected internal virtual bool IsLiveCollection => TCManager?.CurrentCollection == this; + + /// + /// Called periodically by ConnectionHeartbeat. If some part of the repo could not be + /// watched when monitoring started -- for a joiner, the Books folder that Dropbox has + /// not delivered yet -- this is the chance to try again. Does nothing by default. + /// See BL-16729. + /// + protected internal virtual void RetryDeferredWatching() { } + + /// + /// Tell the manager we have noticed we can no longer reach the repo. Goes through the + /// ITeamCollectionManager interface (rather than the concrete TCManager) so it is + /// mockable in tests. + /// + internal void ReportConnectionProblem(TeamCollectionMessage problem) + { + _tcManager?.NoticeConnectionProblem(problem, RepoDescription); + } + /// /// Start monitoring the repo so we can get notifications of new and changed books. /// @@ -456,6 +510,11 @@ protected virtual internal void StartMonitoring() { _monitoring = true; + // The watchers tell us at once if the shared folder is yanked away, but only this + // notices that Dropbox has quietly stopped syncing. See BL-16729. + _heartbeat = new ConnectionHeartbeat(this); + _heartbeat.Start(); + // Set up monitoring for the local folder. Here we are looking for changes // to collection-level files that need to be saved to the repo. // Watching for changes to the repoFolder (or other similar store) are handled @@ -471,11 +530,180 @@ protected virtual internal void StartMonitoring() // Conceivably we should do something to make sure we also see deletions. _localFolderWatcher.NotifyFilter = NotifyFilters.LastWrite; + // The default 8KB buffer holds only a couple of hundred notifications, and this + // watcher covers the whole local collection including every book folder, so it is + // the one most likely to overflow. 64KB is the documented maximum. + _localFolderWatcher.InternalBufferSize = kWatcherBufferSize; + _localFolderWatcher.Changed += OnChanged; _localFolderWatcher.Created += OnChanged; + _localFolderWatcher.Error += OnLocalFolderWatcherError; // Begin watching. - _localFolderWatcher.EnableRaisingEvents = true; + try + { + _localFolderWatcher.EnableRaisingEvents = true; + } + catch (Exception ex) + { + // BL-16679 was a crash from exactly this call. Losing this watcher is not a + // reason to disconnect (see OnLocalFolderWatcherError), but we do want to know. + Logger.WriteError("Could not watch the local collection folder", ex); + NonFatalProblem.ReportSentryOnly(ex, "Could not watch the local collection folder"); + } + } + + /// + /// The buffer FileSystemWatcher uses to hold notifications until we drain them. When it + /// overflows, Windows discards its whole contents and we are simply never told about + /// those changes. 64KB is the documented maximum. + /// + protected const int kWatcherBufferSize = 64 * 1024; + + /// + /// Deliberately NOT a reason to disconnect. This watcher exists only to push local + /// collection-file edits up to the repo; losing it does not stop us seeing our teammates' + /// work. It also watches the user's own disk, so a failure here says nothing about + /// whether the Team Collection is reachable, and a wrong disconnect is the worst outcome + /// we can produce. See BL-16729. + /// + private void OnLocalFolderWatcherError(object sender, ErrorEventArgs e) + { + // As in HandleRepoWatcherError, nothing may escape into the watcher's callback. + try + { + var ex = e.GetException(); + Logger.WriteError("Watcher on the local collection folder failed", ex); + NonFatalProblem.ReportSentryOnly( + ex, + "Watcher on the local collection folder failed" + ); + if (ex is InternalBufferOverflowException) + { + // We lost some notifications of local collection-file edits. Syncing + // collection files to the repo is exactly what we would have done for each of + // them, so just do it once, the same way OnChanged does. + RequestCollectionFilesSyncOnIdle(); + } + } + catch (Exception handlerFailure) + { + NonFatalProblem.ReportSentryOnly(handlerFailure); + } + } + + /// + /// A repo file system watcher failed, so we can no longer see what our teammates do. + /// Raised on a threadpool thread; must not block, and must not dispose the watcher from + /// inside its own callback (NoticeConnectionProblem marshals to the UI thread for us). + /// See BL-16729. + /// + internal void HandleRepoWatcherError(string watchedPath, Exception ex) + { + // Nothing may escape into a FileSystemWatcher callback: an unhandled exception on + // that thread takes the process down with it. + try + { + Logger.WriteError($"Team Collection watcher failed on {watchedPath}", ex); + NonFatalProblem.ReportSentryOnly( + ex, + $"Team Collection watcher failed on {watchedPath}" + ); + + if (ex is InternalBufferOverflowException) + { + HandleLostNotifications(); + return; + } + + // Anything else means the watch itself is dead: .NET will not re-establish it, so + // we would never hear about another change even if the folder came back. We don't + // bother confirming with Directory.Exists first; that would block this thread on a + // dead share and could not change the outcome. + _tcManager?.NoticeConnectionProblem( + new TeamCollectionMessage( + MessageAndMilestoneType.Error, + "TeamCollection.LostContactWithRepo", + "Bloom can no longer watch the Team Collection folder at \"{0}\", so it will not see changes made by your teammates. Usually this means the folder, or the drive or network it is on, is no longer available.", + RepoDescription + ), + RepoDescription + ); + } + catch (Exception handlerFailure) + { + NonFatalProblem.ReportSentryOnly(handlerFailure); + } + } + + /// + /// A watcher's buffer overflowed: the folder is fine, but our buffer filled faster than + /// we drained it, so we lost some notifications. Disconnecting a collection that is + /// actually working would be a self-inflicted outage at exactly the busiest moment. But + /// nothing else about this is visible to the user, so say so twice: an Error message + /// (which makes the Reload Collection button appear) and a toast, since a recoloured + /// button alone is easy to miss. Errors are de-duplicated in the message log, and the + /// toastId de-duplicates the toast, so a storm of overflows yields one of each. + /// + internal void HandleLostNotifications() + { + // Both repo watchers can overflow at the same moment, on different thread-pool + // threads. TeamCollectionMessageLog keeps one unsynchronized list which it + // enumerates to de-duplicate and then appends to, and the UI reads that same list, + // so writing to it from here directly could duplicate entries or throw. Hand the + // work to the UI thread, exactly as the disconnect path does. + Program.RunOnUiThreadLater(() => + { + // By the time this runs, a racing watcher failure may have disconnected us. The + // manager has then swapped in a DisconnectedTeamCollection with its own message + // log, so writing to ours would put the warning somewhere the status dialog no + // longer reads -- and "you may have missed some changes" is moot next to "you + // have lost contact with the collection" anyway. + if (!IsLiveCollection) + return; + MessageLog.WriteMessage( + MessageAndMilestoneType.Error, + kMayHaveMissedChangesId, + kMayHaveMissedChangesEnglish + ); + ToastService.ShowToast( + ToastType.Warning, + text: LocalizationManager.GetString( + kMayHaveMissedChangesId, + kMayHaveMissedChangesEnglish + ), + l10nId: kMayHaveMissedChangesId, + action: new ToastAction { Callback = () => _tcManager?.ShowStatusDialog() }, + toastId: "team-collection-missed-changes" + ); + }); + } + + internal const string kMayHaveMissedChangesId = "TeamCollection.MayHaveMissedChanges"; + private const string kMayHaveMissedChangesEnglish = + "Bloom may have missed some changes your teammates made. Please click \"Reload Collection\" to be sure you have the latest."; + + /// + /// Turn a watcher on, reporting rather than throwing if the OS won't let us watch. + /// Returns false if we did not get monitoring going. Note that HandleRepoWatcherError + /// marshals asynchronously, so a failure here cannot tear the watchers down from inside + /// StartMonitoring itself. + /// + protected internal bool TryStartWatching( + FileSystemWatcherWrapper watcher, + string watchedPath + ) + { + try + { + watcher.EnableRaisingEvents = true; + return true; + } + catch (Exception ex) + { + HandleRepoWatcherError(watchedPath, ex); + return false; + } } private void OnChanged(object sender, FileSystemEventArgs e) @@ -491,10 +719,17 @@ private void OnChanged(object sender, FileSystemEventArgs e) return; // side effect of doing a sync! if (Directory.Exists(e.FullPath)) return; // we seem to get frequent notifications that seem to be spurious for book folders. - // We'll wait for the system to be idle before writing to the repo. This helps to ensure things - // are in a consistent state, as we may get multiple write notifications during the process of - // writing a file. It may also help to ensure that repo writing doesn't interfere somehow with - // whatever is changing things. + RequestCollectionFilesSyncOnIdle(); + } + + /// + /// Arrange to push local collection-file changes to the repo once things are idle. + /// We wait for idle so that things are in a consistent state, as we may get multiple + /// write notifications during the process of writing a file. It may also help to ensure + /// that repo writing doesn't interfere somehow with whatever is changing things. + /// + private void RequestCollectionFilesSyncOnIdle() + { var form = Shell.GetShellOrOtherOpenForm(); // Form.ActiveForm is null when a browser is active if (form != null) { @@ -544,6 +779,8 @@ private void SyncCollectionFilesToRepoOnIdle(object sender, EventArgs e) protected virtual internal void StopMonitoring() { _monitoring = false; + _heartbeat?.Dispose(); + _heartbeat = null; if (_localFolderWatcher != null) { _localFolderWatcher.EnableRaisingEvents = false; @@ -1656,6 +1893,34 @@ public virtual bool HasLocalChangesThatMustBeClobbered(string bookName) return isConflictingCheckedOutStatus || isConflictingCheckSum; } + /// + /// Whether the repo's record of who has this book checked out differs from our local + /// copy's. HasBeenChangedRemotely compares book checksums, and checking a book out or in + /// does not change the book's checksum -- it changes the status stored beside it -- so a + /// checkout made while we were not watching is invisible to that test. The live watcher + /// never needed this, because it reacts to the repo file being written at all; the + /// catch-up scan that runs when we start watching late does. Without it a teammate's + /// checkout stays invisible and the book goes on offering itself for checkout, which is + /// how two people end up editing the same book. See BL-16729. + /// + protected bool HasCheckoutChangedRemotely(string bookName) + { + var repoStatus = GetStatus(bookName); + // Nothing trustworthy to compare against: say no rather than announcing a change we + // cannot substantiate. A repo we cannot read is the watcher's problem, not ours. + if (repoStatus == null || repoStatus.hasInvalidRepoData) + return false; + var localStatus = GetLocalStatus(bookName); + if (localStatus == null) + return false; + // Ordinal comparison, as BookStatus.IsCheckedOutHereBy uses on the same two fields. + // lockedWhen is deliberately not compared: a check-in followed by another checkout + // from the same machine leaves these two the same, and if the book itself changed in + // between the checksum test has already caught it. + return localStatus.lockedBy != repoStatus.lockedBy + || localStatus.lockedWhere != repoStatus.lockedWhere; + } + private bool HasCheckoutConflict(string bookName) { return IsCheckedOutHereBy(GetLocalStatus(bookName)) @@ -2180,7 +2445,7 @@ internal bool IsCheckedOutHereBy(BookStatus status, string email = null) return status.IsCheckedOutHereBy(whoBy); } - bool IsBloomBookFolder(string folderPath) + protected bool IsBloomBookFolder(string folderPath) { return !string.IsNullOrEmpty(BookStorage.FindBookHtmlInFolder(folderPath)); } @@ -2244,6 +2509,24 @@ public void MigrateStatusFiles() /// true if progress messages were reported that are severe enough to warrant /// keeping the progress dialog open until the user responds public bool SyncAtStartup(IWebSocketProgress progress, bool firstTimeJoin = false) + { + try + { + return SyncAtStartupInternal(progress, firstTimeJoin); + } + finally + { + // BL-16729: the real work below clears this on its normal return and on its two + // explicit abort paths, but any other exception -- SynchronizeRepoAndLocal + // catches those and carries on -- used to leave it set for the rest of the + // session. Since IsWritingToRepo consults it, that permanently disabled the + // periodic connection check: Dropbox could stop later and no tick would ever + // look. Clearing it here covers every exit. + _syncIsRunning = false; + } + } + + private bool SyncAtStartupInternal(IWebSocketProgress progress, bool firstTimeJoin) { Debug.Assert( !string.IsNullOrEmpty(CollectionId), @@ -3149,6 +3432,12 @@ protected virtual void Dispose(bool disposing) } Application.Idle -= HandleRemoteBookChangesOnIdle; + + // Ordinary message writes give up quickly on a busy log file, so as not to stall + // the UI thread, and leave what they could not write for the next message. This + // is the last chance for anything still outstanding to reach the file, and by + // now there is no UI left to keep responsive, so it may take its time. + _tcLog.Flush(); } } diff --git a/src/BloomExe/TeamCollection/TeamCollectionApi.cs b/src/BloomExe/TeamCollection/TeamCollectionApi.cs index 496e7cc7407b..9602e1564b69 100644 --- a/src/BloomExe/TeamCollection/TeamCollectionApi.cs +++ b/src/BloomExe/TeamCollection/TeamCollectionApi.cs @@ -251,12 +251,10 @@ private void HandleForceUnlock(ApiRequest request) private void HandleShowStatusDialog(ApiRequest request) { - dynamic messageBundle = new DynamicJson(); - messageBundle.showReloadButton = _tcManager.MessageLog.ShouldShowReloadButton; - _socketServer.LaunchDialog("TeamCollectionDialog", messageBundle); - _tcManager.CurrentCollectionEvenIfDisconnected?.MessageLog.WriteMilestone( - MessageAndMilestoneType.LogDisplayed - ); + // The manager owns this because the toast we raise when we notice a connection + // problem opens the same dialog, and both routes must record the LogDisplayed + // milestone. See BL-16729. + _tcManager.ShowStatusDialog(); request.PostSucceeded(); } diff --git a/src/BloomExe/TeamCollection/TeamCollectionManager.cs b/src/BloomExe/TeamCollection/TeamCollectionManager.cs index 6f734eaa9a7c..ce5336d944f0 100644 --- a/src/BloomExe/TeamCollection/TeamCollectionManager.cs +++ b/src/BloomExe/TeamCollection/TeamCollectionManager.cs @@ -1,10 +1,12 @@ -using System; +using System; using System.IO; using System.Linq; using Bloom.Api; using Bloom.Book; using Bloom.Collection; using Bloom.SubscriptionAndFeatures; +using Bloom.web; +using L10NSharp; using SIL.IO; namespace Bloom.TeamCollection @@ -20,6 +22,21 @@ public interface ITeamCollectionManager CollectionSettings Settings { get; } CollectionLock Lock { get; } bool CheckConnection(); + + /// + /// Something noticed, without the user asking, that we can no longer reach the Team + /// Collection repo: a file system watcher died, or the periodic connection check failed + /// enough times to be believed. Switches to the disconnected state and tells the user. + /// Thread-safe, idempotent, and non-blocking, so it is safe to call from a watcher + /// callback or a background timer. See BL-16729. + /// + void NoticeConnectionProblem(TeamCollectionMessage message, string repoDescription); + + /// + /// Open the Team Collection dialog, as clicking the top-bar Team Collection button does. + /// + void ShowStatusDialog(); + void ConnectToTeamCollection(string repoFolderParentPath, string collectionId); string PlannedRepoFolderPath(string repoFolderParentPath); @@ -59,6 +76,10 @@ public class TeamCollectionManager : IDisposable, ITeamCollectionManager private BookCollectionHolder _bookCollectionHolder; public TeamCollection CurrentCollection { get; private set; } + // The collection we were using before MakeDisconnected replaced it, kept only so that + // Dispose can dispose it. See BL-16729. + private TeamCollection _collectionAwaitingDisposal; + // Normally the same as CurrentCollection, but CurrentCollection is only // non-null when we have a fully functional Team Collection operating. // Sometimes a TC may be disconnected, that is, we know this is a TC, @@ -453,16 +474,160 @@ public bool CheckConnection() if (connectionProblem != null) { - MakeDisconnected(connectionProblem, CurrentCollection.RepoDescription); + MakeDisconnected(connectionProblem, CurrentCollection?.RepoDescription); return false; } return true; } - public void MakeDisconnected(TeamCollectionMessage message, string repoDescription) + /// + public void NoticeConnectionProblem(TeamCollectionMessage message, string repoDescription) + { + if (CurrentCollection == null) + { + // Already disconnected, or not a TC at all -- but also the window during + // ConnectToTeamCollection where a brand-new collection is being set up and has + // not been published as CurrentCollection yet. A watcher failing to start in + // that window has nowhere to go, so at least record it rather than dropping it + // on the floor. See the open question on BL-16729 about whether a collection we + // cannot watch should be treated as disconnected outright. + SIL.Reporting.Logger.WriteError( + "Team Collection connection problem with no current collection to disconnect: " + + message?.TextForDisplay, + new ApplicationException(message?.RawEnglishMessageTemplate ?? "unknown") + ); + NonFatalProblem.ReportSentryOnly( + $"Team Collection connection problem dropped (no current collection): {message?.L10NId}" + ); + return; + } + + // Deliberately no "a disconnect is already queued" flag here. Racing callers are + // handled by MakeDisconnected being idempotent: whichever gets to the UI thread + // first does the work and returns true, the rest return false and do nothing. A + // flag would have to be cleared by the queued delegate, and a delegate that never + // runs (the form's handle is destroyed before it is dispatched) would then latch + // us into a state where we could never disconnect again. + Program.RunOnUiThreadLater(() => + { + try + { + if (!MakeDisconnected(message, repoDescription)) + return; // someone else already disconnected us; don't toast twice. + // Recolouring the Team Collection button is too easy to miss for something + // this consequential, so put a notification in front of the user as well. + // No durationSeconds, so it stays until they close or click it. + ToastService.ShowToast( + ToastType.Error, + text: LocalizationManager.GetString( + kNoLongerSeeingChangesId, + kNoLongerSeeingChangesEnglish + ), + l10nId: kNoLongerSeeingChangesId, + action: new ToastAction { Callback = ShowStatusDialog }, + toastId: "team-collection-disconnected" + ); + } + catch (Exception ex) + { + NonFatalProblem.ReportSentryOnly(ex); + } + }); + } + + private const string kNoLongerSeeingChangesId = "TeamCollection.NoLongerSeeingChanges"; + private const string kNoLongerSeeingChangesEnglish = + "Bloom can no longer see the Team Collection folder, so you will not see changes made by your teammates."; + + /// + public void ShowStatusDialog() + { + dynamic messageBundle = new DynamicJson(); + messageBundle.showReloadButton = MessageLog.ShouldShowReloadButton; + _webSocketServer.LaunchDialog("TeamCollectionDialog", messageBundle); + CurrentCollectionEvenIfDisconnected?.MessageLog.WriteMilestone( + MessageAndMilestoneType.LogDisplayed + ); + } + + /// + /// Switch to the disconnected state, recording the given message. Returns false, having + /// done nothing, if we were already disconnected -- racing callers (two watchers failing + /// at once, a heartbeat and a checkout) must not each write another copy of the messages. + /// + public bool MakeDisconnected(TeamCollectionMessage message, string repoDescription) + { + TeamCollection previousCollection; + // Claim the transition atomically. Callers arrive both directly (a synchronous + // CheckConnection from an API handler that is not on the UI thread) and indirectly + // (a watcher or heartbeat failure marshalled onto the UI thread), so without this + // two of them could capture the same live collection, both pass the guard, and both + // go on to stop it and build a replacement. + lock (_disconnectLock) + { + if (_disconnectInProgress) + return false; + previousCollection = CurrentCollection; + if ( + previousCollection == null + && CurrentCollectionEvenIfDisconnected is DisconnectedTeamCollection + ) + { + return false; + } + // Null this inside the claim, so that anything looking at it while we build the + // replacement sees "disconnected" rather than a half-built state. + CurrentCollection = null; + _disconnectInProgress = true; + } + try + { + CompleteDisconnect(previousCollection, message, repoDescription); + } + finally + { + // Cleared in a finally, and only ever within this one synchronous method, so + // unlike a flag held across an asynchronous marshal it cannot latch on and leave + // us permanently unable to disconnect. + lock (_disconnectLock) + { + _disconnectInProgress = false; + } + } + return true; + } + + private readonly object _disconnectLock = new object(); + private bool _disconnectInProgress; + + /// + /// The rest of the disconnect, run by whichever caller won the claim above. Deliberately + /// outside the lock: it writes to the message log, which raises an event that reaches + /// WinForms and the websocket server. + /// + private void CompleteDisconnect( + TeamCollection previousCollection, + TeamCollectionMessage message, + string repoDescription + ) { - CurrentCollection = null; + // BL-16729: we have given up on this collection, so stop its file system watchers and + // its periodic connection check. Otherwise a dead watcher goes on raising Error, and a + // live one goes on queueing repo changes into an object nobody is using any more. + try + { + previousCollection?.StopMonitoring(); + } + catch (Exception ex) + { + NonFatalProblem.ReportSentryOnly(ex); + } + // Keep it so Dispose can still clean it up; nothing else refers to it now. If we + // already had one (disconnected, reconnected, disconnected again), that one has + // certainly finished with, so let it go rather than leaking it. + _collectionAwaitingDisposal?.Dispose(); + _collectionAwaitingDisposal = previousCollection; // This will show the TC icon in error state, and if the dialog is shown it will have this one message. CurrentCollectionEvenIfDisconnected = new DisconnectedTeamCollection( this, @@ -471,6 +636,18 @@ public void MakeDisconnected(TeamCollectionMessage message, string repoDescripti ); CurrentCollectionEvenIfDisconnected.SocketServer = SocketServer; CurrentCollectionEvenIfDisconnected.TCManager = this; + // BL-16729: the stand-in answers book-status questions out of the local status + // files, and DisconnectedTeamCollection.GetBookStatusJsonFromRepo only trusts a + // local status whose collectionId matches this collection's. SetCollectionId runs + // once, when the collection is opened (WorkspaceModel), so before we could + // disconnect part way through a session the stand-in was always built early enough + // for that call to reach it. Now that we can, a stand-in built mid-session would + // keep a null CollectionId, every book's local status would look like it belonged + // to some other collection, and the whole collection would present as newly created + // local books -- no checkout status, nothing shown as being in the repo. Carry the + // id across from the collection we are replacing. + CurrentCollectionEvenIfDisconnected.CollectionId = + previousCollection?.CollectionId ?? Settings?.CollectionId; // Every call to MessageLog.WriteMessage() also raises the TeamCollectionStatusChanged event. CurrentCollectionEvenIfDisconnected.MessageLog.WriteMessage(message); CurrentCollectionEvenIfDisconnected.MessageLog.WriteMessage( @@ -560,6 +737,18 @@ public string PlannedRepoFolderPath(string repoFolderParentPath) public void Dispose() { CurrentCollection?.Dispose(); + // After a disconnect, CurrentCollection is null and the DisconnectedTeamCollection + // that stands in for it has a message log of its own -- the one holding the "you are + // now disconnected" messages. Nothing else disposes that object, so without this its + // log would never be flushed, and any message whose quick append had failed would be + // lost at shutdown (BL-16729). While we are connected the two are the same object, so + // only dispose it when it is a different one. + if (!ReferenceEquals(CurrentCollectionEvenIfDisconnected, CurrentCollection)) + CurrentCollectionEvenIfDisconnected?.Dispose(); + // A collection we disconnected from part way through the session. MakeDisconnected + // has already stopped its watchers, but it still holds the objects (BL-16729). + _collectionAwaitingDisposal?.Dispose(); + _collectionAwaitingDisposal = null; } public void RaiseBookStatusChanged(BookStatusChangeEventArgs eventInfo) diff --git a/src/BloomExe/TeamCollection/TeamCollectionMessageLog.cs b/src/BloomExe/TeamCollection/TeamCollectionMessageLog.cs index d5dcecb1325e..e2423c3da3ca 100644 --- a/src/BloomExe/TeamCollection/TeamCollectionMessageLog.cs +++ b/src/BloomExe/TeamCollection/TeamCollectionMessageLog.cs @@ -1,7 +1,9 @@ using System; +using System.Collections.Concurrent; using System.Collections.Generic; using System.IO; using System.Linq; +using System.Text; using Bloom.web; using SIL.Code; using SIL.IO; @@ -39,6 +41,7 @@ void WriteMessage( ); void WriteMessage(TeamCollectionMessage message); void WriteMilestone(MessageAndMilestoneType milestoneType); + void Flush(); BloomWebSocketProgressEvent[] GetProgressMessages(); } @@ -61,29 +64,104 @@ public TeamCollectionMessageLog(string logFilePath) { _oldMessageLength = new FileInfo(_logFilePath).Length; } - Messages = new List(); } // Review: currently includes milestones. Should it? - public List Messages { get; private set; } + private readonly List _messages = new List(); + + /// + /// A snapshot of the messages so far. Deliberately a copy: callers enumerate this from + /// threads that are not the UI thread (teamCollection/getLog and + /// teamCollection/logImportant are both registered with handleOnUiThread false), and + /// enumerating the live list while another thread appends to it throws + /// "Collection was modified". See BL-16729. + /// + public List Messages + { + get + { + lock (_messagesLock) + { + return new List(_messages); + } + } + } + + /// + /// Guards Messages. Writers are not all on the UI thread: several API endpoints are + /// registered with handleOnUiThread false and can end up here via CheckConnection, and + /// (BL-16729) a file system watcher failing calls in from a thread-pool thread. Because + /// WriteMessage enumerates the list to de-duplicate and then appends to it, while the + /// status properties below enumerate the same list, an unguarded overlap could produce + /// duplicate entries or throw InvalidOperationException. + /// + private readonly object _messagesLock = new object(); + + /// + /// Guards the log file. Deliberately NOT _messagesLock: appending is normally a matter + /// of microseconds, but when something else has the file open (antivirus, a backup + /// agent, the read-only collection folder of BL-16772) RobustFile retries over a period + /// of seconds, and for all that time nothing may read Messages, TeamCollectionStatus and + /// the rest -- TeamCollectionStatus is read on the UI thread whenever the Team + /// Collection button refreshes. + /// + private readonly object _fileLock = new object(); + + /// + /// Lines waiting to be appended to the log file. Enqueued while _messagesLock is held, + /// so their order is exactly the order of _messages, and dequeued only while _fileLock + /// is held, so one thread writes at a time. Between them the file ends up in the same + /// order as the in-memory list, without the file being touched under _messagesLock. + /// + private readonly ConcurrentQueue _pendingWrites = new ConcurrentQueue(); + + /// + /// Lines an append failed to write, kept in order and written ahead of everything else + /// by the next attempt. Guarded by _fileLock. + /// + private List _carryOver = new List(); + + // How hard an ordinary append tries. This often runs on the UI thread (every remote + // change is handled from Application.Idle), so we must not stall it for seconds the way + // RobustFile would: a couple of quick tries catches the usual transient sharing + // violation, and anything still unwritten waits in _carryOver for the next message, or + // for the stubborn Flush at shutdown. + private const int kQuickAppendAttempts = 2; + private const int kQuickAppendRetryMs = 50; + + // Only IOException is worth a second try: that is the transient case, something else + // holding the file open for a moment. A permissions failure (the read-only collection + // folder of BL-16772) will not get better while we wait, so let it out at once. + private static readonly ISet kTransientAppendExceptions = new HashSet + { + typeof(IOException), + }; + + // A ceiling on _carryOver, so that a log file that stays unwritable for the rest of the + // session cannot grow it without bound. Generous: the messages are short, and this is + // only reached when something is badly wrong. + private const int kMaxCarryOverChars = 64 * 1024; public List CurrentErrors { get { - // correctly 0 if none match - var index = - Messages.FindLastIndex(m => - m.MessageType == MessageAndMilestoneType.LogDisplayed - || m.MessageType == MessageAndMilestoneType.Reloaded - ) + 1; - return Messages - .Skip(index) - .Where(m => - m.MessageType == MessageAndMilestoneType.Error - || m.MessageType == MessageAndMilestoneType.ErrorNoReload - ) - .ToList(); + lock (_messagesLock) + { + // correctly 0 if none match + var index = + _messages.FindLastIndex(m => + m.MessageType == MessageAndMilestoneType.LogDisplayed + || m.MessageType == MessageAndMilestoneType.Reloaded + ) + 1; + return _messages + .Skip(index) + .Where(m => + m.MessageType == MessageAndMilestoneType.Error + || m.MessageType == MessageAndMilestoneType.ErrorNoReload + ) + .ToList(); + } } } @@ -101,17 +179,21 @@ public List ReloadMessages { get { - // correctly 0 if none match - var index = - Messages.FindLastIndex(m => m.MessageType == MessageAndMilestoneType.Reloaded) - + 1; - return Messages - .Skip(index) - .Where(m => - m.MessageType == MessageAndMilestoneType.Error - || m.MessageType == MessageAndMilestoneType.NewStuff - ) - .ToList(); + lock (_messagesLock) + { + // correctly 0 if none match + var index = + _messages.FindLastIndex(m => + m.MessageType == MessageAndMilestoneType.Reloaded + ) + 1; + return _messages + .Skip(index) + .Where(m => + m.MessageType == MessageAndMilestoneType.Error + || m.MessageType == MessageAndMilestoneType.NewStuff + ) + .ToList(); + } } } @@ -127,14 +209,18 @@ public List CurrentNewStuff { get { - // correctly 0 if none match - var index = - Messages.FindLastIndex(m => m.MessageType == MessageAndMilestoneType.Reloaded) - + 1; - return Messages - .Skip(index) - .Where(m => m.MessageType == MessageAndMilestoneType.NewStuff) - .ToList(); + lock (_messagesLock) + { + // correctly 0 if none match + var index = + _messages.FindLastIndex(m => + m.MessageType == MessageAndMilestoneType.Reloaded + ) + 1; + return _messages + .Skip(index) + .Where(m => m.MessageType == MessageAndMilestoneType.NewStuff) + .ToList(); + } } } @@ -142,12 +228,17 @@ public TeamCollectionMessage CurrentClobberMessage { get { - var last = Messages.FindLast(m => - m.MessageType == MessageAndMilestoneType.ClobberPending - || m.MessageType == MessageAndMilestoneType.ShowedClobbered - || m.MessageType == MessageAndMilestoneType.Reloaded - ); - return last?.MessageType == MessageAndMilestoneType.ClobberPending ? last : null; + lock (_messagesLock) + { + var last = _messages.FindLast(m => + m.MessageType == MessageAndMilestoneType.ClobberPending + || m.MessageType == MessageAndMilestoneType.ShowedClobbered + || m.MessageType == MessageAndMilestoneType.Reloaded + ); + return last?.MessageType == MessageAndMilestoneType.ClobberPending + ? last + : null; + } } } @@ -155,10 +246,13 @@ public DateTime LastReloadTime { get { - var last = Messages.FindLast(m => - m.MessageType == MessageAndMilestoneType.Reloaded - ); - return last == null ? DateTime.MinValue : last.When; + lock (_messagesLock) + { + var last = _messages.FindLast(m => + m.MessageType == MessageAndMilestoneType.Reloaded + ); + return last == null ? DateTime.MinValue : last.When; + } } } @@ -184,39 +278,169 @@ public void WriteMessage( string param1 = "" ) { - if (IsRedundantMessage(messageType, l10nId, message, param0, param1)) - return; var msg = new TeamCollectionMessage(messageType, l10nId, message, param0, param1); - WriteMessage(msg); + // The de-duplication check and the append have to be one atomic step, or two + // concurrent writers can both decide the message is new and both add it. + lock (_messagesLock) + { + if (IsRedundantMessage(messageType, l10nId, message, param0, param1)) + return; + _messages.Add(msg); + QueueForPersisting(msg); + } + DrainPendingWrites(); + AfterMessageAdded(msg); } public void WriteMessage(TeamCollectionMessage message) { - Messages.Add(message); - SIL.Reporting.Logger.WriteEvent(message.TextForDisplay); - TeamCollectionManager.RaiseTeamCollectionStatusChanged(); + lock (_messagesLock) + { + _messages.Add(message); + QueueForPersisting(message); + } + DrainPendingWrites(); + AfterMessageAdded(message); + } + + /// + /// Line up one message to be appended to the log file. Called with _messagesLock held, + /// which is what makes the queue's order the same as the in-memory list's. + /// + private void QueueForPersisting(TeamCollectionMessage message) + { // Using Environment.NewLine here means the format of the file will be appropriate for the // computer we are running on. It's possible a shared collection might be used by both // Linux and Windows. But that's OK, because .NET line reading accepts either line // break on either platform. - var toPersist = message.ToPersistedForm + Environment.NewLine; + _pendingWrites.Enqueue(message.ToPersistedForm + Environment.NewLine); + } + + /// + /// Append everything queued so far to the log file, giving up quickly if the file is + /// busy. Deliberately called with _messagesLock released, so a slow append cannot block + /// the readers; _fileLock instead keeps two threads from colliding over the file, and + /// because the queue is FIFO and is only ever drained under that lock, the file stays in + /// the same order as _messages. + /// + /// + /// Nothing can be left behind unwritten: whoever queues a message goes on to call this, + /// and waits for _fileLock rather than giving up, so its line is written either by this + /// call or by the drain that is already running. Finding nothing to write, which is what + /// happens when that other drain took our line, is therefore a normal outcome. + /// + private void DrainPendingWrites() + { + WritePendingMessages(beStubborn: false); + } + + /// + /// Write anything still waiting, trying as hard as RobustFile does. Ordinary appends give + /// up quickly so as not to stall the thread writing the message -- usually the UI thread + /// -- and leave what they could not write for the next message to carry out. At shutdown + /// there is no next message, so this is those lines' last chance to reach the file, and + /// here it is worth waiting out whatever has the file open. + /// + public void Flush() + { + WritePendingMessages(beStubborn: true); + } + + private void WritePendingMessages(bool beStubborn) + { + lock (_fileLock) + { + // Whatever an earlier attempt could not write goes first, so that the file stays + // in the order of the in-memory list. A burst of messages (SyncAtStartup + // produces one) thus costs a single append rather than one per message. + var lines = _carryOver; + _carryOver = new List(); + while (_pendingWrites.TryDequeue(out var line)) + lines.Add(line); + if (lines.Count == 0) + return; + if (TryAppend(string.Concat(lines), beStubborn)) + return; + // The messages are already in Messages, so the current session still shows them, + // and AfterMessageAdded writes them to the ordinary log as well; the worst case + // is that they don't survive a restart. Keep them for the next attempt. + _carryOver = lines; + TrimCarryOverIfTooBig(); + } + } + + /// + /// Called with _fileLock held. + /// + /// true if the text reached the file. + private bool TryAppend(string text, bool beStubborn) + { try { - RobustFile.AppendAllText(_logFilePath, toPersist); + if (beStubborn) + { + RobustFile.AppendAllText(_logFilePath, text); + } + else + { + // What RobustFile.AppendAllText does -- the same call, and so the same + // encoding (UTF-8, no BOM), so the two paths can append to one file + // interchangeably -- but over milliseconds rather than seconds. + RobustFileIO.QuickAppendAllText( + _logFilePath, + text, + kQuickAppendAttempts, + kQuickAppendRetryMs, + kTransientAppendExceptions + ); + } + return true; } catch (Exception ex) { - // The message is already in Messages (so the current session still shows it) and in - // the Logger above; it just won't survive a restart. Not being able to write it must - // not take Bloom down: this very method is called while reporting a TC initialization - // failure, and when the underlying problem is an unwritable collection folder - // (e.g. read-only files, BL-16772), throwing here turned a degraded-but-working - // Team Collection into a collection that could not open at all. + // Not being able to write must not take Bloom down: this path is used while + // reporting a TC initialization failure, and when the underlying problem is an + // unwritable collection folder (e.g. read-only files, BL-16772), throwing here + // turned a degraded-but-working Team Collection into a collection that could not + // open at all. SIL.Reporting.Logger.WriteError( - $"Could not persist Team Collection message to {_logFilePath}", + $"Could not persist Team Collection messages to {_logFilePath}", ex ); + return false; + } + } + + /// + /// Called with _fileLock held. Drops the oldest unwritten lines if they have piled up, + /// keeping the most recent, which are the ones most likely to explain what went wrong. + /// + private void TrimCarryOverIfTooBig() + { + var total = _carryOver.Sum(line => line.Length); + if (total <= kMaxCarryOverChars) + return; + var toDrop = 0; + while (toDrop < _carryOver.Count && total > kMaxCarryOverChars) + { + total -= _carryOver[toDrop].Length; + toDrop++; } + _carryOver.RemoveRange(0, toDrop); + SIL.Reporting.Logger.WriteEvent( + $"Gave up on {toDrop} Team Collection messages that could not be written to {_logFilePath}" + ); + } + + /// + /// Deliberately called with the lock released: raising the status-changed event reaches + /// WinForms and the websocket server, and holding a lock across that is how deadlocks + /// happen. Everything here reads only the message it was handed. + /// + private void AfterMessageAdded(TeamCollectionMessage message) + { + SIL.Reporting.Logger.WriteEvent(message.TextForDisplay); + TeamCollectionManager.RaiseTeamCollectionStatusChanged(); } private bool MatchParams(string p1, string p2) @@ -226,6 +450,34 @@ private bool MatchParams(string p1, string p2) return p1 == p2; } + /// + /// Errors whose de-duplication starts afresh after a Reload. Repeating most errors is + /// just noise -- a bad zip file in the repo produces the same complaint over and over -- + /// which is why the check below otherwise looks at the whole session. But "Bloom may have + /// missed some changes, please click Reload Collection" is different: once the user has + /// acted on it, a fresh occurrence is telling them something genuinely new, and swallowing + /// it leaves them believing they are up to date when Bloom has just decided they are not. + /// See BL-16729. + /// + private static readonly ISet kErrorsRepeatableAfterReload = new HashSet + { + TeamCollection.kMayHaveMissedChangesId, + }; + + /// + /// The index just past the most recent Reloaded milestone, or 0 if there has not been one + /// this session. Call with _messagesLock held, as IsRedundantMessage's caller does. + /// + private int IndexAfterLastReloadedMilestone() + { + for (var i = _messages.Count - 1; i >= 0; i--) + { + if (_messages[i].MessageType == MessageAndMilestoneType.Reloaded) + return i + 1; + } + return 0; + } + private bool IsRedundantMessage( MessageAndMilestoneType messageType, string l10nId, @@ -254,16 +506,26 @@ string param1 // the message is redundant with a current session report. But currently we reset completely for each // session, and problems (particularly the one produced by a bad zip file in the repo) tend to be very // frequent. We need to look at everything to weed out duplicates. - return Messages.Any(msg => - ( - msg.MessageType == MessageAndMilestoneType.Error - || msg.MessageType == MessageAndMilestoneType.ErrorNoReload + // ...except for the few messages a Reload answers; see kErrorsRepeatableAfterReload. + var searchFrom = kErrorsRepeatableAfterReload.Contains(l10nId) + ? IndexAfterLastReloadedMilestone() + : 0; + for (var i = searchFrom; i < _messages.Count; i++) + { + var msg = _messages[i]; + if ( + ( + msg.MessageType == MessageAndMilestoneType.Error + || msg.MessageType == MessageAndMilestoneType.ErrorNoReload + ) + && msg.L10NId == l10nId + && msg.RawEnglishMessageTemplate == message + && MatchParams(msg.Param0, param0) + && MatchParams(msg.Param1, param1) ) - && msg.L10NId == l10nId - && msg.RawEnglishMessageTemplate == message - && MatchParams(msg.Param0, param0) - && MatchParams(msg.Param1, param1) - ); + return true; + } + return false; } return false; } diff --git a/src/BloomTests/TeamCollection/ConnectionFailureTrackerTests.cs b/src/BloomTests/TeamCollection/ConnectionFailureTrackerTests.cs new file mode 100644 index 000000000000..71f818dfd4e8 --- /dev/null +++ b/src/BloomTests/TeamCollection/ConnectionFailureTrackerTests.cs @@ -0,0 +1,109 @@ +using Bloom.TeamCollection; +using NUnit.Framework; + +namespace BloomTests.TeamCollection +{ + /// + /// Tests for the "wait for a second failure before believing it" policy that keeps a + /// transient network blip from disconnecting a Team Collection that is actually working. + /// See BL-16729. + /// + [TestFixture] + public class ConnectionFailureTrackerTests + { + private ConnectionFailureTracker _tracker; + + [SetUp] + public void Setup() + { + _tracker = new ConnectionFailureTracker(); + } + + private static TeamCollectionMessage Problem(string l10nId) + { + return new TeamCollectionMessage( + MessageAndMilestoneType.Error, + l10nId, + "some English text" + ); + } + + [Test] + public void RecordResult_Success_ReportsNoProblem() + { + Assert.That(_tracker.RecordResult(null), Is.False); + } + + [Test] + public void RecordResult_SingleFailure_DoesNotYetConclude() + { + Assert.That( + _tracker.RecordResult(Problem("TeamCollection.NoNetwork")), + Is.False, + "one failure should not be enough; it is very often a transient blip" + ); + } + + [Test] + public void RecordResult_TwoSameFailures_Concludes() + { + // Sanity check that the first one really did not conclude. + Assert.That(_tracker.RecordResult(Problem("TeamCollection.NoNetwork")), Is.False); + + Assert.That(_tracker.RecordResult(Problem("TeamCollection.NoNetwork")), Is.True); + } + + [Test] + public void RecordResult_SuccessBetweenFailures_StartsOver() + { + Assert.That(_tracker.RecordResult(Problem("TeamCollection.NoNetwork")), Is.False); + Assert.That(_tracker.RecordResult(null), Is.False); + + Assert.That( + _tracker.RecordResult(Problem("TeamCollection.NoNetwork")), + Is.False, + "the intervening success means these two failures were not consecutive" + ); + } + + [Test] + public void RecordResult_DifferentKindsOfFailure_StartsOver() + { + Assert.That(_tracker.RecordResult(Problem("TeamCollection.NoNetwork")), Is.False); + + Assert.That( + _tracker.RecordResult(Problem("TeamCollection.MissingRepo")), + Is.False, + "two different problems are two transients, not one sustained outage" + ); + // ...but two of the second kind in a row still counts. + Assert.That(_tracker.RecordResult(Problem("TeamCollection.MissingRepo")), Is.True); + } + + [Test] + public void Reset_ForgetsPreviousFailure() + { + Assert.That(_tracker.RecordResult(Problem("TeamCollection.NoNetwork")), Is.False); + + _tracker.Reset(); + + Assert.That( + _tracker.RecordResult(Problem("TeamCollection.NoNetwork")), + Is.False, + "Reset should have discarded the earlier failure" + ); + } + + [Test] + public void RecordResult_ManyFailures_KeepsConcluding() + { + _tracker.RecordResult(Problem("TeamCollection.NoNetwork")); + Assert.That(_tracker.RecordResult(Problem("TeamCollection.NoNetwork")), Is.True); + Assert.That( + _tracker.RecordResult(Problem("TeamCollection.NoNetwork")), + Is.True, + "a continuing outage should keep reporting as one" + ); + } + } +} diff --git a/src/BloomTests/TeamCollection/FolderTeamCollectionTests2.cs b/src/BloomTests/TeamCollection/FolderTeamCollectionTests2.cs index fad18c4a0d37..e64493f7ffa6 100644 --- a/src/BloomTests/TeamCollection/FolderTeamCollectionTests2.cs +++ b/src/BloomTests/TeamCollection/FolderTeamCollectionTests2.cs @@ -1,4 +1,5 @@ using System; +using System.Collections.Generic; using System.IO; using System.Linq; using System.Threading; @@ -7,8 +8,10 @@ using Bloom.Collection; using Bloom.MiscUI; using Bloom.TeamCollection; +using Bloom.Utils; using Bloom.web; using BloomTemp; +using BloomTests.DataBuilders; using Moq; using NUnit.Framework; using SIL.IO; @@ -1772,5 +1775,1362 @@ public void HandleCollectionSettingsChange_NoRepoSettings_LeavesMinimumVersionAl } ); } + + #region BL-16729: noticing that we can no longer watch the repo + + /// + /// Callers reach MakeDisconnected both directly (a synchronous CheckConnection from an + /// API handler that is not on the UI thread) and indirectly (a watcher or heartbeat + /// failure marshalled onto the UI thread), so two can arrive at once. Exactly one must + /// do the work. + /// + [Test] + public void MakeDisconnected_ManyThreadsAtOnce_DisconnectsExactlyOnce() + { + using (var collectionFolder = new TemporaryFolder("DisconnectRace_Collection")) + using (var repoFolder = new TemporaryFolder("DisconnectRace_Repo")) + { + Directory.CreateDirectory(Path.Combine(repoFolder.FolderPath, "Books")); + var settingsPath = CollectionSettings.GetDefaultSettingsFilePath( + collectionFolder.FolderPath + ); + RobustFile.WriteAllText(settingsPath, "This is a fake settings file"); + FolderTeamCollection.CreateTeamCollectionLinkFile( + collectionFolder.FolderPath, + repoFolder.FolderPath + ); + using ( + var tcManager = new TeamCollectionManager( + settingsPath, + null, + new BookStatusChangeEvent(), + null, + null, + null + ) + ) + { + Assert.That( + tcManager.CurrentCollection, + Is.Not.Null, + "setup problem: should have connected to the repo" + ); + + var wonCount = 0; + System.Threading.Tasks.Parallel.For( + 0, + 16, + i => + { + var won = tcManager.MakeDisconnected( + new TeamCollectionMessage( + MessageAndMilestoneType.Error, + "TeamCollection.LostContactWithRepo", + "we lost it" + ), + repoFolder.FolderPath + ); + if (won) + System.Threading.Interlocked.Increment(ref wonCount); + } + ); + + Assert.That( + wonCount, + Is.EqualTo(1), + "exactly one caller should be told it did the disconnecting" + ); + // Counting messages alone would not catch a double disconnect, because the + // message log de-duplicates Errors -- so assert on the object identity too. + Assert.That( + tcManager.CurrentCollectionEvenIfDisconnected, + Is.InstanceOf() + ); + Assert.That( + tcManager.MessageLog.Messages.Count(m => + m.L10NId == "TeamCollection.OperatingDisconnected" + ), + Is.EqualTo(1), + "the disconnect messages should have been written exactly once" + ); + } + } + } + + /// + /// Makes a TeamCollection over throwaway folders, with a mocked manager, and runs the + /// given check against it. Used by the watcher-failure tests below, which do not need + /// any book content. + /// + private void WithMockManagedCollection( + string testName, + Action> check + ) + { + using (var collectionFolder = new TemporaryFolder(testName + "_Collection")) + using (var repoFolder = new TemporaryFolder(testName + "_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + check(tc, mockTcManager); + } + } + } + + [Test] + public void HandleRepoWatcherError_WatchIsDead_NoticesConnectionProblem() + { + WithMockManagedCollection( + "WatcherErrorDisconnects", + (tc, mockTcManager) => + { + // sut: the sort of exception the OS raises when the share goes away. + tc.HandleRepoWatcherError( + Path.Combine(tc.RepoDescription, "Books"), + new IOException("The specified network name is no longer available") + ); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.Is(msg => + msg.L10NId == "TeamCollection.LostContactWithRepo" + ), + It.IsAny() + ), + Times.Once, + "a dead watch means we can no longer see our teammates' work, so we should disconnect" + ); + } + ); + } + + /// + /// A buffer overflow means we lost some notifications, but the folder is perfectly + /// reachable. Disconnecting would be a self-inflicted outage at the busiest moment. + /// + [Test] + public void HandleRepoWatcherError_BufferOverflow_WarnsButDoesNotDisconnect() + { + WithMockManagedCollection( + "WatcherOverflowWarns", + (tc, mockTcManager) => + { + tc.PretendIsLiveCollection = true; + Assert.That( + tc.MessageLog.CurrentErrors, + Is.Empty, + "setup problem: should start with no errors in the log" + ); + + // sut + tc.HandleRepoWatcherError( + Path.Combine(tc.RepoDescription, "Books"), + new InternalBufferOverflowException() + ); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never, + "the connection is fine; we only lost some notifications" + ); + var errors = tc.MessageLog.CurrentErrors; + Assert.That(errors.Count, Is.EqualTo(1)); + Assert.That( + errors[0].L10NId, + Is.EqualTo("TeamCollection.MayHaveMissedChanges") + ); + Assert.That( + tc.MessageLog.ShouldShowReloadButton, + Is.True, + "the user needs the Reload Collection button to catch up on what we missed" + ); + } + ); + } + + [Test] + public void HandleRepoWatcherError_RepeatedOverflow_AddsOnlyOneMessage() + { + WithMockManagedCollection( + "WatcherOverflowDedupes", + (tc, mockTcManager) => + { + tc.PretendIsLiveCollection = true; + var booksPath = Path.Combine(tc.RepoDescription, "Books"); + + tc.HandleRepoWatcherError(booksPath, new InternalBufferOverflowException()); + Assert.That( + tc.MessageLog.CurrentErrors.Count, + Is.EqualTo(1), + "setup problem: the first overflow should have logged one error" + ); + + // sut + tc.HandleRepoWatcherError(booksPath, new InternalBufferOverflowException()); + + Assert.That( + tc.MessageLog.CurrentErrors.Count, + Is.EqualTo(1), + "a storm of overflows should not fill the log with identical messages" + ); + } + ); + } + + /// + /// A joiner's Dropbox may not have delivered the repo's Books folder by the time Bloom + /// starts watching. That used to mean no book watcher for the rest of the session -- + /// and, because of an early return, no Other watcher either. See BL-16729. + /// + [Test] + public void StartMonitoring_NoBooksFolderYet_StillWatchesTheOtherFolder() + { + using (var collectionFolder = new TemporaryFolder("NoBooksYet_Collection")) + using (var repoFolder = new TemporaryFolder("NoBooksYet_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + var settingsPath = CollectionSettings.GetDefaultSettingsFilePath( + collectionFolder.FolderPath + ); + Directory.CreateDirectory(Path.GetDirectoryName(settingsPath)); + File.WriteAllText(settingsPath, "This is the initial value"); + // Creates the repo's Other folder, but deliberately NOT Books. + tc.CopyRepoCollectionFilesFromLocal(collectionFolder.FolderPath); + Assert.That( + Directory.Exists(Path.Combine(repoFolder.FolderPath, "Books")), + Is.False, + "setup problem: this test is about the Books folder being absent" + ); + + tc.SetupMonitoringBehavior(); + var collectionChangedRaised = new ManualResetEvent(false); + EventHandler monitorFunction = (sender, args) => + collectionChangedRaised.Set(); + tc.RepoCollectionFilesChanged += monitorFunction; + + // sut: change a collection file, which only the Other watcher can see + Thread.Sleep(10); + RobustFile.WriteAllText( + FolderTeamCollection.GetRepoProjectFilesZipPath(repoFolder.FolderPath), + @"This is changed" + ); + var raised = collectionChangedRaised.WaitOne(1000); + + tc.RepoCollectionFilesChanged -= monitorFunction; + tc.StopMonitoring(); + + Assert.That( + raised, + Is.True, + "a missing Books folder should not stop us watching collection files" + ); + } + } + } + + /// + /// Once Dropbox delivers the Books folder, the periodic check should start watching it + /// and announce whatever is already sitting there -- from the watcher's point of view + /// those books are all new since Bloom started. See BL-16729. + /// + [Test] + public void RetryDeferredWatching_BooksFolderHasArrived_AnnouncesTheBooksAlreadyInIt() + { + using (var collectionFolder = new TemporaryFolder("BooksArrive_Collection")) + using (var repoFolder = new TemporaryFolder("BooksArrive_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + tc.PretendIsLiveCollection = true; + tc.StartMonitoring(); // no Books folder yet, so watching is deferred + + var newBooks = new List(); + EventHandler handler = (sender, args) => + newBooks.Add(args.BookFileName); + tc.NewBook += handler; + var deletedBooks = new List(); + EventHandler deleteHandler = (sender, args) => + deletedBooks.Add(args.BookFileName); + tc.DeleteRepoBookFile += deleteHandler; + try + { + // Nothing to find while the folder is still absent. + tc.RetryDeferredWatching(); + Assert.That( + newBooks, + Is.Empty, + "setup problem: there is no Books folder yet, so nothing to announce" + ); + + // Dropbox delivers the folder, with a book already in it. + var booksPath = Path.Combine(repoFolder.FolderPath, "Books"); + Directory.CreateDirectory(booksPath); + RobustFile.WriteAllText( + Path.Combine(booksPath, "Arrived book.bloom"), + "not really a zip" + ); + + // sut + tc.RetryDeferredWatching(); + + Assert.That(newBooks, Is.EqualTo(new[] { "Arrived book.bloom" })); + Assert.That( + deletedBooks, + Is.Empty, + "there are no local books, so nothing can have been deleted" + ); + + // And it is idempotent: a second tick must not announce it again. + newBooks.Clear(); + tc.RetryDeferredWatching(); + Assert.That( + newBooks, + Is.Empty, + "having started watching, later ticks should do nothing" + ); + } + finally + { + tc.NewBook -= handler; + tc.DeleteRepoBookFile -= deleteHandler; + tc.StopMonitoring(); + } + } + } + } + + /// + /// A book deleted from the repo while we had no watcher raised no Deleted event either, + /// so nothing has told us our local copy is obsolete. Once the Books folder arrives we + /// must raise that event ourselves -- and, exactly as for an event from the watcher, the + /// local book must only really go away if the repo has a tombstone showing that someone + /// deliberately deleted it. See BL-16729. + /// + [TestCase(true)] + [TestCase(false)] + public void RetryDeferredWatching_BookDeletedWhileNotWatching_AnnouncedAndRemovedOnlyIfTombstoned( + bool makeTombstone + ) + { + using (var collectionFolder = new TemporaryFolder("BookDeletedWhileNotWatching_Coll")) + using (var repoFolder = new TemporaryFolder("BookDeletedWhileNotWatching_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + // Two books that really are in the collection, both checked in, so each has + // a genuine local status file and a genuine .bloom in the repo. + var keptPath = MakeLocalBook(collectionFolder.FolderPath, "Kept book"); + var doomedPath = MakeLocalBook(collectionFolder.FolderPath, "Doomed book"); + tc.PutBook(keptPath); + tc.PutBook(doomedPath); + + // A teammate deletes one of them. It is the notification of this, not the + // deletion itself, that never reaches us. + tc.DeleteBookFromRepo(doomedPath, makeTombstone); + + // Dropbox has not delivered the Books folder to this machine yet. (The + // tombstone lives at the repo root, so it is not stashed with it.) + var booksPath = Path.Combine(repoFolder.FolderPath, "Books"); + var stashPath = Path.Combine(repoFolder.FolderPath, "StashedBooks"); + Directory.Move(booksPath, stashPath); + + tc.PretendIsLiveCollection = true; + tc.StartMonitoring(); // no Books folder, so watching is deferred + + var deletedBooks = new List(); + EventHandler deleteHandler = (sender, args) => + deletedBooks.Add(args.BookFileName); + tc.DeleteRepoBookFile += deleteHandler; + try + { + tc.RetryDeferredWatching(); + Assert.That( + deletedBooks, + Is.Empty, + "setup problem: there is no Books folder yet, so nothing to announce" + ); + + // Dropbox delivers the folder, now without the deleted book. + Directory.Move(stashPath, booksPath); + + // sut + tc.RetryDeferredWatching(); + + Assert.That( + deletedBooks, + Is.EqualTo(new[] { "Doomed book.bloom" }), + "the book missing from the repo should be announced as deleted, and " + + "the one still there should not be" + ); + + // Follow the announcement through to what it actually does. (Only the + // five-second pause before this is untestable, which is why the logic + // lives in HandleDeletedRepoFile.) + tc.HandleDeletedRepoFile("Doomed book.bloom"); + Assert.That( + Directory.Exists(doomedPath), + Is.EqualTo(!makeTombstone), + makeTombstone + ? "a tombstoned deletion should remove the local book" + : "without a tombstone we have no evidence of a deliberate " + + "deletion, so the local book must survive" + ); + Assert.That( + Directory.Exists(keptPath), + Is.True, + "the book still in the repo must not be touched" + ); + } + finally + { + tc.DeleteRepoBookFile -= deleteHandler; + tc.StopMonitoring(); + } + } + } + } + + /// + /// A local book that has no counterpart in the repo is not necessarily a deleted one: + /// it may never have been checked in, or it may be checked out and renamed here, in + /// which case the repo still has it under its old name. Neither should be announced as + /// deleted when the Books folder finally arrives. See BL-16729. + /// + [Test] + public void RetryDeferredWatching_BooksWithNoRepoCounterpart_NotAnnouncedAsDeleted() + { + using (var collectionFolder = new TemporaryFolder("NoRepoCounterpart_Coll")) + using (var repoFolder = new TemporaryFolder("NoRepoCounterpart_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + // A book created here and never checked in. Deliberately it has no local + // status file; that is how we tell it from one deleted remotely. + var localOnlyPath = MakeLocalBook( + collectionFolder.FolderPath, + "Local only book" + ); + + // A book checked out here and renamed, but not yet checked in, so the repo + // still holds it under its old name. + var originalPath = MakeLocalBook(collectionFolder.FolderPath, "My book"); + tc.PutBook(originalPath); + tc.AttemptLock("My book", "fred@nowhere.org"); + var renamedPath = Path.Combine(collectionFolder.FolderPath, "Renamed book"); + Directory.Move(originalPath, renamedPath); + tc.HandleBookRename("My book", "Renamed book"); + Assert.That( + tc.GetLocalStatus("Renamed book").oldName, + Is.EqualTo("My book"), + "setup problem: the local rename should have recorded the repo's name" + ); + + // Dropbox has not delivered the Books folder to this machine yet. + var booksPath = Path.Combine(repoFolder.FolderPath, "Books"); + var stashPath = Path.Combine(repoFolder.FolderPath, "StashedBooks"); + Directory.Move(booksPath, stashPath); + + tc.PretendIsLiveCollection = true; + tc.StartMonitoring(); // no Books folder, so watching is deferred + + var deletedBooks = new List(); + EventHandler deleteHandler = (sender, args) => + deletedBooks.Add(args.BookFileName); + tc.DeleteRepoBookFile += deleteHandler; + try + { + // Dropbox delivers the folder, holding only "My book.bloom". + Directory.Move(stashPath, booksPath); + + // sut + tc.RetryDeferredWatching(); + + Assert.That( + deletedBooks, + Is.Empty, + "neither a never-checked-in book nor one renamed here has lost a " + + "repo counterpart" + ); + Assert.That( + Directory.Exists(localOnlyPath) && Directory.Exists(renamedPath), + Is.True, + "sanity check: neither local book should have been touched" + ); + } + finally + { + tc.DeleteRepoBookFile -= deleteHandler; + tc.StopMonitoring(); + } + } + } + } + + /// + /// Writes a minimal but real book folder (an htm, so it counts as a Bloom book, and a + /// meta.json, so it has an ID that tombstones can be keyed on) into the collection. + /// + /// The path to the book folder. + /// + /// The catch-up scan that runs when we start watching late asks HasBeenChangedRemotely, + /// which compares book checksums -- and a checkout does not change a book's checksum, it + /// changes the status stored beside it. So a teammate's checkout made while we were not + /// watching was invisible, and the book went on offering itself for checkout. See BL-16729. + /// + [Test] + public void HasCheckoutChangedRemotely_TeammateCheckedItOutWhileWeWereNotWatching_True() + { + using (var collectionFolder = new TemporaryFolder("CheckoutChanged_Collection")) + using (var repoFolder = new TemporaryFolder("CheckoutChanged_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + var bookPath = MakeLocalBook(collectionFolder.FolderPath, "Shared book"); + tc.PutBook(bookPath); + Assert.That( + tc.CallHasCheckoutChangedRemotely("Shared book"), + Is.False, + "setup problem: freshly checked in, the repo and local status should agree" + ); + + // A teammate checks it out. Only the repo's copy of the status records that; + // nothing has told us, because we were not watching. + var repoStatus = tc.GetStatus("Shared book"); + tc.WriteRepoStatusOnly( + "Shared book", + repoStatus.WithLockedBy("fred@nowhere.org", "Fred", "Smith") + ); + + // sut + Assert.That( + tc.CallHasCheckoutChangedRemotely("Shared book"), + Is.True, + "a checkout recorded only in the repo is exactly the change this has to see" + ); + + // The book's content has not changed, so the checksum test alone would have + // missed it -- which is the whole reason this test exists. + Assert.That( + tc.HasBeenChangedRemotely("Shared book"), + Is.False, + "sanity check: the checksum is unchanged, so the old test finds nothing" + ); + } + } + } + + /// + /// The other direction: a teammate checking a book back in is also a status-only change, + /// and must not be mistaken for "nothing happened" either. + /// + [Test] + public void HasCheckoutChangedRemotely_TeammateCheckedItBackIn_True() + { + using (var collectionFolder = new TemporaryFolder("CheckinChanged_Collection")) + using (var repoFolder = new TemporaryFolder("CheckinChanged_Repo")) + { + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath + ) + ) + { + var bookPath = MakeLocalBook(collectionFolder.FolderPath, "Shared book"); + tc.PutBook(bookPath); + // Both copies agree that a teammate has it out. + var checkedOut = tc.GetStatus("Shared book") + .WithLockedBy("fred@nowhere.org", "Fred", "Smith"); + // WriteBookStatus writes both copies, so after this they agree. + tc.WriteBookStatus("Shared book", checkedOut); + Assert.That( + tc.CallHasCheckoutChangedRemotely("Shared book"), + Is.False, + "setup problem: both copies should agree that Fred has it out" + ); + + // Fred checks it back in; again only the repo knows. + tc.WriteRepoStatusOnly("Shared book", checkedOut.WithLockedBy(null)); + + // sut + Assert.That( + tc.CallHasCheckoutChangedRemotely("Shared book"), + Is.True, + "a check-in recorded only in the repo is a change we have to notice too" + ); + } + } + } + + private static string MakeLocalBook(string collectionFolderPath, string title) + { + return new BookFolderBuilder() + .WithRootFolder(collectionFolderPath) + .WithTitle(title) + .WithHtm("") + .Build() + .BuiltBookFolderPath; + } + + [Test] + public void RetryDeferredWatching_BooksFolderWasThereAllAlong_DoesNothing() + { + WithMockManagedCollection( + "BooksWereThere", + (tc, mockTcManager) => + { + var booksPath = Path.Combine(tc.RepoDescription, "Books"); + Directory.CreateDirectory(booksPath); + RobustFile.WriteAllText( + Path.Combine(booksPath, "Existing book.bloom"), + "not really a zip" + ); + tc.PretendIsLiveCollection = true; + tc.StartMonitoring(); // watcher starts normally; nothing is deferred + + var newBooks = new List(); + EventHandler handler = (sender, args) => + newBooks.Add(args.BookFileName); + tc.NewBook += handler; + try + { + // sut + tc.RetryDeferredWatching(); + + Assert.That( + newBooks, + Is.Empty, + "a collection that was watched from the start must not have its " + + "existing books re-announced every minute" + ); + } + finally + { + tc.NewBook -= handler; + tc.StopMonitoring(); + } + } + ); + } + + /// + /// A sync that throws must not leave the collection looking permanently busy: the + /// periodic connection check skips its tick while a write is in progress, so a stuck + /// flag would silently disable it for the rest of the session. + /// + [Test] + public void SyncAtStartup_Throws_StillClearsTheBusyFlag() + { + WithMockManagedCollection( + "SyncThrowsClearsBusy", + (tc, mockTcManager) => + { + Assert.That( + tc.IsWritingToRepo, + Is.False, + "setup problem: should not start out busy" + ); + + // The repo has no Books folder and no collection files, so the sync throws + // somewhere inside rather than returning normally. + Assert.Catch( + () => tc.SyncAtStartup(new ProgressSpy(), firstTimeJoin: false), + "setup problem: this sync was supposed to fail" + ); + + Assert.That( + tc.IsWritingToRepo, + Is.False, + "a failed sync must not leave the collection looking busy forever" + ); + } + ); + } + + /// + /// If a racing watcher failure disconnected us while the overflow warning was queued for + /// the UI thread, the manager has swapped in a different collection with a different + /// message log -- so writing ours would put the warning where nothing reads it. + /// + [Test] + public void HandleRepoWatcherError_OverflowAfterDisconnect_WritesNothing() + { + WithMockManagedCollection( + "WatcherOverflowAfterDisconnect", + (tc, mockTcManager) => + { + tc.PretendIsLiveCollection = false; // we have already been disconnected + + tc.HandleRepoWatcherError( + Path.Combine(tc.RepoDescription, "Books"), + new InternalBufferOverflowException() + ); + + Assert.That( + tc.MessageLog.CurrentErrors, + Is.Empty, + "nothing should go into the log of a collection we have given up on" + ); + } + ); + } + + /// + /// BL-16679 was a crash from EnableRaisingEvents throwing. Now we report instead. + /// A watcher with no Path set is the deterministic way to make it throw. + /// + [Test] + public void TryStartWatching_WatcherCannotStart_ReportsRatherThanThrowing() + { + WithMockManagedCollection( + "TryStartWatchingFails", + (tc, mockTcManager) => + { + using (var watcher = new FileSystemWatcherWrapper()) // no Path: cannot start + { + // Sanity check that this really is a watcher that cannot be started. + // (Currently a FileNotFoundException, but the point of TryStartWatching + // is that we don't care which exception it is.) + Assert.Catch( + () => watcher.EnableRaisingEvents = true, + "setup problem: a path-less watcher was supposed to refuse to start" + ); + + // sut + bool started = true; + Assert.DoesNotThrow(() => + started = tc.TryStartWatching(watcher, "some path") + ); + + Assert.That(started, Is.False); + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Once, + "failing to start watching the repo means we cannot see our teammates' work" + ); + } + } + ); + } + + /// + /// The whole StartMonitoring path over a repo that isn't there. Note this currently + /// stops at the "no Books folder" early return, so it is a smoke test rather than a + /// test of the EnableRaisingEvents guard (see TryStartWatching_... above for that). + /// + [Test] + public void StartMonitoring_RepoFolderDoesNotExist_DoesNotThrow() + { + using (var collectionFolder = new TemporaryFolder("StartMonitoringNoRepo_Collection")) + { + var missingRepoPath = Path.Combine( + collectionFolder.FolderPath, + "no such repo folder" + ); + Assert.That( + Directory.Exists(missingRepoPath), + Is.False, + "setup problem: the repo folder was supposed to be missing" + ); + var mockTcManager = new Mock(); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + missingRepoPath + ) + ) + { + // sut + Assert.DoesNotThrow(() => tc.StartMonitoring()); + Assert.DoesNotThrow(() => tc.StopMonitoring()); + } + } + } + + [Test] + public void StartAndStopMonitoring_TracksIsMonitoring() + { + WithMockManagedCollection( + "IsMonitoringTracks", + (tc, mockTcManager) => + { + Directory.CreateDirectory(Path.Combine(tc.RepoDescription, "Books")); + Assert.That(tc.IsMonitoring, Is.False, "setup problem: not started yet"); + + tc.StartMonitoring(); + Assert.That(tc.IsMonitoring, Is.True); + + tc.StopMonitoring(); + Assert.That(tc.IsMonitoring, Is.False); + } + ); + } + + /// + /// The periodic check calls CheckConnection many times over a session. History messages + /// are not de-duplicated, so a probe that wrote them would fill up log.txt and raise a + /// status-changed event on every tick of a perfectly healthy session. + /// + /// Note the limits of this test: the two History writes it guards live behind + /// "the repo is inside a Dropbox folder AND Dropbox is unreachable AND the folder is + /// also shared on the LAN", which a temp folder cannot reproduce. So this catches a + /// probe that writes unconditionally, but not a regression confined to that branch -- + /// for which the `if (writeHistoryMessages)` guards themselves are the evidence. + /// + [Test] + public void CheckConnection_QuietProbe_WritesNoMessages() + { + using (var collectionFolder = new TemporaryFolder("QuietProbe_Collection")) + using (var repoFolder = new TemporaryFolder("QuietProbe_Repo")) + { + var log = new TeamCollectionMessageLog( + TeamCollectionManager.GetTcLogPathFromLcPath(collectionFolder.FolderPath) + ); + var mockTcManager = new Mock(); + mockTcManager.Setup(m => m.MessageLog).Returns(log); + using ( + var tc = new TestFolderTeamCollection( + mockTcManager.Object, + collectionFolder.FolderPath, + repoFolder.FolderPath, + log + ) + ) + { + Assert.That( + log.Messages, + Is.Empty, + "setup problem: should start with an empty log" + ); + + // sut + for (var i = 0; i < 5; i++) + tc.CheckConnection(writeHistoryMessages: false); + + Assert.That( + log.Messages, + Is.Empty, + "a quiet probe must not write to the message log, however often it runs" + ); + } + } + } + + /// + /// Once we give up on a collection, its watchers must stop; otherwise a dead one goes on + /// raising Error and a live one goes on queueing changes into an object nobody uses. + /// Dispose must still be able to reach it. See BL-16729. + /// + [Test] + public void MakeDisconnected_StopsOldCollectionAndSwitchesState() + { + using (var collectionFolder = new TemporaryFolder("MakeDisconnected_Collection")) + using (var repoFolder = new TemporaryFolder("MakeDisconnected_Repo")) + { + Directory.CreateDirectory(Path.Combine(repoFolder.FolderPath, "Books")); + var settingsPath = CollectionSettings.GetDefaultSettingsFilePath( + collectionFolder.FolderPath + ); + RobustFile.WriteAllText(settingsPath, "This is a fake settings file"); + FolderTeamCollection.CreateTeamCollectionLinkFile( + collectionFolder.FolderPath, + repoFolder.FolderPath + ); + using ( + var tcManager = new TeamCollectionManager( + settingsPath, + null, + new BookStatusChangeEvent(), + null, + null, + null + ) + ) + { + var originalCollection = tcManager.CurrentCollection; + Assert.That( + originalCollection, + Is.Not.Null, + "setup problem: should have connected to the repo" + ); + originalCollection.StartMonitoring(); + Assert.That( + originalCollection.IsMonitoring, + Is.True, + "setup problem: should be monitoring before we disconnect" + ); + + // sut + tcManager.MakeDisconnected( + new TeamCollectionMessage( + MessageAndMilestoneType.Error, + "TeamCollection.LostContactWithRepo", + "we lost it" + ), + repoFolder.FolderPath + ); + + Assert.That(tcManager.CurrentCollection, Is.Null); + Assert.That( + tcManager.CurrentCollectionEvenIfDisconnected, + Is.InstanceOf() + ); + Assert.That( + originalCollection.IsMonitoring, + Is.False, + "the collection we gave up on should have stopped watching" + ); + Assert.That(tcManager.MessageLog.ShouldShowReloadButton, Is.True); + } + } + } + + [Test] + public void MakeDisconnected_CalledTwice_DoesNotWriteMessagesTwice() + { + using (var collectionFolder = new TemporaryFolder("DisconnectTwice_Collection")) + using (var repoFolder = new TemporaryFolder("DisconnectTwice_Repo")) + { + Directory.CreateDirectory(Path.Combine(repoFolder.FolderPath, "Books")); + var settingsPath = CollectionSettings.GetDefaultSettingsFilePath( + collectionFolder.FolderPath + ); + RobustFile.WriteAllText(settingsPath, "This is a fake settings file"); + FolderTeamCollection.CreateTeamCollectionLinkFile( + collectionFolder.FolderPath, + repoFolder.FolderPath + ); + using ( + var tcManager = new TeamCollectionManager( + settingsPath, + null, + new BookStatusChangeEvent(), + null, + null, + null + ) + ) + { + Assert.That( + tcManager.CurrentCollection, + Is.Not.Null, + "setup problem: should have connected to the repo" + ); + var message = new TeamCollectionMessage( + MessageAndMilestoneType.Error, + "TeamCollection.LostContactWithRepo", + "we lost it" + ); + Assert.That( + tcManager.MakeDisconnected(message, repoFolder.FolderPath), + Is.True, + "setup problem: the first call should have done the disconnecting" + ); + var disconnectedCollection = tcManager.CurrentCollectionEvenIfDisconnected; + var messageCountAfterFirst = tcManager.MessageLog.Messages.Count; + + // sut: a second watcher failing, or the heartbeat, arriving right behind the first. + Assert.That( + tcManager.MakeDisconnected(message, repoFolder.FolderPath), + Is.False, + "the second caller must be told it did nothing, so it does not also toast" + ); + + Assert.That( + tcManager.CurrentCollectionEvenIfDisconnected, + Is.SameAs(disconnectedCollection), + "should not have built a second DisconnectedTeamCollection" + ); + Assert.That( + tcManager.MessageLog.Messages.Count, + Is.EqualTo(messageCountAfterFirst), + "should not have written another copy of the disconnect messages" + ); + } + } + } + + /// + /// Drives ConnectionHeartbeat.UpdateTcConnectionStatus directly -- no timer, no network, no real repo -- so + /// the confirm-then-act policy, the guards, and disposal are all covered. Devin flagged + /// this integration as unverified on PR #8338. + /// + private void WithHeartbeat( + string testName, + Action< + ConnectionHeartbeat, + TestFolderTeamCollection, + Mock + > check + ) + { + WithMockManagedCollection( + testName, + (tc, mockTcManager) => + { + Directory.CreateDirectory(Path.Combine(tc.RepoDescription, "Books")); + tc.InterceptCheckConnection = true; + tc.PretendIsLiveCollection = true; + tc.StartMonitoring(); // so IsMonitoring is true + try + { + Assert.That( + tc.IsMonitoring, + Is.True, + "setup problem: the heartbeat's guard needs monitoring to be on" + ); + check(new ConnectionHeartbeat(tc), tc, mockTcManager); + } + finally + { + tc.StopMonitoring(); + } + } + ); + } + + private static TeamCollectionMessage AProblem(string l10nId = "TeamCollection.NoNetwork") + { + return new TeamCollectionMessage( + MessageAndMilestoneType.Error, + l10nId, + "something is wrong" + ); + } + + [Test] + public void HeartbeatTick_ConnectionFine_ChecksAndDoesNothing() + { + WithHeartbeat( + "HeartbeatOk", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = null; + + heartbeat.UpdateTcConnectionStatus(null); + + Assert.That( + tc.CheckConnectionCallCount, + Is.EqualTo(1), + "it should actually have looked" + ); + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never + ); + } + ); + } + + [Test] + public void HeartbeatTick_OneFailure_WaitsForConfirmationBeforeDisconnecting() + { + WithHeartbeat( + "HeartbeatOneFailure", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + + heartbeat.UpdateTcConnectionStatus(null); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never, + "one failed check is very often a transient blip; we must confirm first" + ); + } + ); + } + + [Test] + public void HeartbeatTick_TwoFailuresInARow_Disconnects() + { + WithHeartbeat( + "HeartbeatTwoFailures", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + + heartbeat.UpdateTcConnectionStatus(null); + heartbeat.UpdateTcConnectionStatus(null); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.Is(msg => + msg.L10NId == "TeamCollection.NoNetwork" + ), + It.IsAny() + ), + Times.Once + ); + } + ); + } + + [Test] + public void HeartbeatTick_RecoveryBetweenFailures_DoesNotDisconnect() + { + WithHeartbeat( + "HeartbeatRecovers", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + heartbeat.UpdateTcConnectionStatus(null); + tc.PretendConnectionProblem = null; + heartbeat.UpdateTcConnectionStatus(null); + tc.PretendConnectionProblem = AProblem(); + + heartbeat.UpdateTcConnectionStatus(null); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never, + "the good check in between means those two failures were not consecutive" + ); + } + ); + } + + [Test] + public void HeartbeatTick_WritingToRepo_SkipsTheCheckAndForgetsEarlierFailures() + { + WithHeartbeat( + "HeartbeatBusy", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + heartbeat.UpdateTcConnectionStatus(null); // one failure on the record + var callsBefore = tc.CheckConnectionCallCount; + + tc.PretendIsWritingToRepo = true; + heartbeat.UpdateTcConnectionStatus(null); + + Assert.That( + tc.CheckConnectionCallCount, + Is.EqualTo(callsBefore), + "should not even look while a check-in or sync is writing to the repo" + ); + + // The skipped tick breaks the run, so the next failure starts over. + tc.PretendIsWritingToRepo = false; + heartbeat.UpdateTcConnectionStatus(null); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never, + "a failure either side of a skipped tick is not a consecutive run" + ); + } + ); + } + + /// + /// A probe that throws tells us nothing either way, so it must break the run rather than + /// silently preserving the earlier strike. Otherwise a failure, a throwing probe, and + /// another failure would disconnect a collection that was never shown to be unreachable + /// twice in a row. + /// + [Test] + public void HeartbeatTick_ProbeThrowsBetweenFailures_DoesNotDisconnect() + { + WithHeartbeat( + "HeartbeatProbeThrows", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + heartbeat.UpdateTcConnectionStatus(null); // strike one + + tc.PretendCheckConnectionThrows = true; + Assert.DoesNotThrow( + () => heartbeat.UpdateTcConnectionStatus(null), + "a throwing probe must not escape the tick" + ); + tc.PretendCheckConnectionThrows = false; + + heartbeat.UpdateTcConnectionStatus(null); + + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never, + "the failures either side of the throwing probe were not consecutive" + ); + + // Sanity check that the tracker is merely reset, not broken: two clean + // failures in a row after this should still disconnect. + heartbeat.UpdateTcConnectionStatus(null); + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Once + ); + } + ); + } + + [Test] + public void HeartbeatTick_NotTheLiveCollection_SkipsTheCheck() + { + WithHeartbeat( + "HeartbeatNotLive", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + tc.PretendIsLiveCollection = false; // e.g. we already disconnected from it + + heartbeat.UpdateTcConnectionStatus(null); + heartbeat.UpdateTcConnectionStatus(null); + + Assert.That(tc.CheckConnectionCallCount, Is.EqualTo(0)); + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never + ); + } + ); + } + + [Test] + public void HeartbeatTick_AfterDispose_DoesNothing() + { + WithHeartbeat( + "HeartbeatDisposed", + (heartbeat, tc, mockTcManager) => + { + tc.PretendConnectionProblem = AProblem(); + heartbeat.Dispose(); + + // Timer.Dispose does not wait for a callback already under way, so a UpdateTcConnectionStatus + // can still arrive after this point. It must be inert. + Assert.DoesNotThrow(() => heartbeat.UpdateTcConnectionStatus(null)); + Assert.DoesNotThrow(() => heartbeat.UpdateTcConnectionStatus(null)); + + Assert.That(tc.CheckConnectionCallCount, Is.EqualTo(0)); + mockTcManager.Verify( + m => + m.NoticeConnectionProblem( + It.IsAny(), + It.IsAny() + ), + Times.Never + ); + } + ); + } + + [Test] + public void HeartbeatStart_UnderUnitTests_DoesNotStartATimer() + { + WithHeartbeat( + "HeartbeatNoTimerInTests", + (heartbeat, tc, mockTcManager) => + { + Assert.That( + Program.RunningUnitTests, + Is.True, + "setup problem: this test is about the RunningUnitTests guard" + ); + tc.PretendConnectionProblem = AProblem(); + + heartbeat.Start(); + + // Assert on the timer itself, not on how many probes have happened: the + // first tick would be IntervalMs (a minute) away, so a probe count is zero + // whether the guard is there or not, and this test would pass even with the + // guard deleted -- while leaking the very timer it exists to rule out. + Assert.That( + heartbeat.IsTimerArmed, + Is.False, + "unit test runs must not be left with thread-pool timers probing the real network" + ); + Assert.That(tc.CheckConnectionCallCount, Is.EqualTo(0)); + heartbeat.Dispose(); + } + ); + } + + #endregion } } diff --git a/src/BloomTests/TeamCollection/TeamCollectionManagerTests.cs b/src/BloomTests/TeamCollection/TeamCollectionManagerTests.cs index c120ec2ea157..4e272175f579 100644 --- a/src/BloomTests/TeamCollection/TeamCollectionManagerTests.cs +++ b/src/BloomTests/TeamCollection/TeamCollectionManagerTests.cs @@ -14,6 +14,89 @@ namespace BloomTests.TeamCollection { public class TeamCollectionManagerTests { + /// + /// When we disconnect part way through a session, CurrentCollection becomes null and the + /// DisconnectedTeamCollection standing in for it carries a message log of its own -- the + /// one holding the "you are now disconnected" messages. Since an ordinary append gives up + /// quickly on a busy log file and leaves the message to be written later, shutting down + /// has to flush that log too, or those messages are lost. See BL-16729. + /// + [Test] + public void Dispose_DisconnectedMidSession_FlushesTheDisconnectedCollectionsLog() + { + using (var collectionFolder = new TemporaryFolder("FlushDisconnectedLog_Collection")) + using (var sharedFolder = new TemporaryFolder("FlushDisconnectedLog_Shared")) + { + var settingsPath = Path.Combine( + collectionFolder.FolderPath, + Path.GetFileName(collectionFolder.FolderPath) + ".bloomCollection" + ); + RobustFile.WriteAllText(settingsPath, ""); + FolderTeamCollection.CreateTeamCollectionLinkFile( + collectionFolder.FolderPath, + sharedFolder.FolderPath + ); + var tcManager = new TeamCollectionManager( + settingsPath, + null, + new BookStatusChangeEvent(), + null, + null, + null + ); + Assert.That( + tcManager.CurrentCollection, + Is.Not.Null, + "setup problem: it should have connected, or there is nothing to disconnect" + ); + + var logPath = TeamCollectionManager.GetTcLogPathFromLcPath( + collectionFolder.FolderPath + ); + // Disconnect while nothing can write to the log file, so the disconnect messages + // have to wait in the message log instead of reaching the file. + using ( + new FileStream(logPath, FileMode.OpenOrCreate, FileAccess.Write, FileShare.None) + ) + { + tcManager.MakeDisconnected( + new TeamCollectionMessage( + MessageAndMilestoneType.Error, + "TeamCollection.NoNetwork", + "No network is available on this computer." + ), + "test repo" + ); + } + Assert.That( + tcManager.CurrentCollection, + Is.Null, + "setup problem: it should now be disconnected" + ); + Assert.That( + tcManager.CurrentCollectionEvenIfDisconnected.MessageLog.Messages.Any(m => + m.L10NId == "TeamCollection.OperatingDisconnected" + ), + Is.True, + "setup problem: the disconnect messages should be in memory" + ); + Assert.That( + RobustFile.ReadAllText(logPath), + Does.Not.Contain("OperatingDisconnected"), + "setup problem: the busy file should not have received them yet" + ); + + // sut + tcManager.Dispose(); + + Assert.That( + RobustFile.ReadAllText(logPath), + Does.Contain("OperatingDisconnected"), + "disposing the manager should have flushed the disconnected collection's log" + ); + } + } + /// /// A collection whose name ends with a period gets a folder without it, because Windows drops /// trailing periods when it creates a folder, so its settings file name does not match its @@ -77,5 +160,79 @@ public void OkToEditCollectionSettings_SettingsFileNameDoesNotMatchFolderName_Su Assert.That(tcManager.OkToEditCollectionSettings, Is.True); } } + + /// + /// The DisconnectedTeamCollection that stands in after a disconnect answers book-status + /// questions from the local status files, and only trusts a local status whose + /// collectionId matches its own CollectionId. Nothing sets that id after the collection + /// has been opened, so a stand-in built part way through a session must inherit it from + /// the collection it replaces; otherwise every book looks like a new local book that was + /// never in the repo. See BL-16729. + /// + [Test] + public void MakeDisconnected_MidSession_StandInInheritsTheCollectionId() + { + using (var collectionFolder = new TemporaryFolder("DisconnectKeepsId_Collection")) + using (var sharedFolder = new TemporaryFolder("DisconnectKeepsId_Shared")) + { + var settingsPath = Path.Combine( + collectionFolder.FolderPath, + Path.GetFileName(collectionFolder.FolderPath) + ".bloomCollection" + ); + RobustFile.WriteAllText(settingsPath, ""); + FolderTeamCollection.CreateTeamCollectionLinkFile( + collectionFolder.FolderPath, + sharedFolder.FolderPath + ); + var tcManager = new TeamCollectionManager( + settingsPath, + null, + new BookStatusChangeEvent(), + null, + null, + null + ); + Assert.That( + tcManager.CurrentCollection, + Is.Not.Null, + "setup problem: it should have connected, or there is nothing to disconnect" + ); + // What WorkspaceModel does once, when the collection is opened. + var collectionId = Bloom.TeamCollection.TeamCollection.GenerateCollectionId(); + tcManager.SetCollectionId(collectionId); + Assert.That( + tcManager.CurrentCollection.CollectionId, + Is.EqualTo(collectionId), + "setup problem: the live collection should be carrying the id we are about to lose" + ); + + // sut + tcManager.MakeDisconnected( + new TeamCollectionMessage( + MessageAndMilestoneType.Error, + "TeamCollection.NoNetwork", + "No network is available on this computer." + ), + "test repo" + ); + + Assert.That( + tcManager.CurrentCollection, + Is.Null, + "setup problem: it should now be disconnected" + ); + Assert.That( + tcManager.CurrentCollectionEvenIfDisconnected, + Is.TypeOf(), + "setup problem: a stand-in should have replaced the live collection" + ); + Assert.That( + tcManager.CurrentCollectionEvenIfDisconnected.CollectionId, + Is.EqualTo(collectionId), + "the stand-in must keep the collection id, or every book's local status looks " + + "like it belongs to a different collection" + ); + } + } } } diff --git a/src/BloomTests/TeamCollection/TeamCollectionMessageLogTests.cs b/src/BloomTests/TeamCollection/TeamCollectionMessageLogTests.cs index 6be9838d5874..e3d014cf683d 100644 --- a/src/BloomTests/TeamCollection/TeamCollectionMessageLogTests.cs +++ b/src/BloomTests/TeamCollection/TeamCollectionMessageLogTests.cs @@ -6,6 +6,8 @@ using System.Threading.Tasks; using Bloom.TeamCollection; using Bloom.web; +using BloomTemp; +using Moq; using NUnit.Framework; using SIL.IO; @@ -394,6 +396,205 @@ public void ReadCurrentNewStuff_ReloadMilestones_RecoversNewStuff() Assert.That(messages, Has.Count.EqualTo(2)); } + /// + /// Writers are not all on the UI thread, and the file is no longer appended to under the + /// lock that guards the in-memory list, so this pins down the thing that buys: the file + /// must still end up in exactly the order of the in-memory log, with nothing lost. + /// + [Test] + public void WriteMessage_ManyThreadsAtOnce_FileIsInTheSameOrderAsTheMessages() + { + const int threadCount = 8; + const int messagesPerThread = 25; + // History messages are never treated as redundant, so every one of these must be + // kept; and each has distinct parameters, so the order of the file is meaningful. + var writers = Enumerable + .Range(0, threadCount) + .Select(thread => + Task.Run(() => + { + for (var i = 0; i < messagesPerThread; i++) + _messageLog.WriteMessage( + MessageAndMilestoneType.History, + "TeamCollection.CheckoutMsg", + "{0} checked out the book {1}", + $"user{thread}@somewhere.org", + // Non-ASCII, so that this also catches the two append paths + // (File.AppendAllText and RobustFile's) disagreeing about + // encoding, which would leave the file unreadable. + $"Böök {i} — ñ" + ); + }) + ) + .ToArray(); + Assert.That( + Task.WaitAll(writers, TimeSpan.FromMinutes(1)), + Is.True, + "the writers should all have finished" + ); + + var expected = _messageLog.Messages.Select(m => m.ToPersistedForm).ToList(); + Assert.That( + expected, + Has.Count.EqualTo(threadCount * messagesPerThread), + "sanity check: every message should have been kept in memory" + ); + Assert.That( + expected.Distinct().Count(), + Is.EqualTo(expected.Count), + "sanity check: the messages must be distinguishable, or this proves nothing " + + "about their order" + ); + + Assert.That( + ReadPersistedLines(), + Is.EqualTo(expected), + "the log file should hold every message, in the order of the in-memory log" + ); + } + + /// + /// An ordinary append gives up quickly rather than stall the thread writing the message, + /// which is usually the UI thread. What it could not write must not be lost, and must not + /// jump ahead of later messages when it is finally written. + /// + [Test] + public void WriteMessage_LogFileBusy_CarriedOverAndWrittenInOrderByTheNextMessage() + { + using ( + new FileStream( + _logFile.Path, + FileMode.Open, + FileAccess.Write, + FileShare.None // nothing else can write while we hold this + ) + ) + { + MakeHistory1(); + } + Assert.That( + ReadPersistedLines(), + Is.Empty, + "setup problem: the message should not have reached the file while it was held" + ); + + // A later message, written when the file is free again. + MakeHistory2(); + + var expected = _messageLog.Messages.Select(m => m.ToPersistedForm).ToList(); + Assert.That( + expected, + Has.Count.EqualTo(2), + "sanity check: both messages should be in memory whatever the file did" + ); + Assert.That( + ReadPersistedLines(), + Is.EqualTo(expected), + "the carried-over message should have been written, ahead of the later one" + ); + } + + [Test] + public void Flush_MessageCouldNotBeWrittenEarlier_WritesIt() + { + using (new FileStream(_logFile.Path, FileMode.Open, FileAccess.Write, FileShare.None)) + { + MakeHistory1(); + } + Assert.That( + ReadPersistedLines(), + Is.Empty, + "setup problem: the message should not have reached the file while it was held" + ); + + // sut: at shutdown there is no next message to carry it out, so this is its last chance. + _messageLog.Flush(); + + Assert.That( + ReadPersistedLines(), + Is.EqualTo(_messageLog.Messages.Select(m => m.ToPersistedForm).ToList()), + "Flush should have written the message the busy file rejected" + ); + } + + [Test] + public void Flush_NothingOutstanding_WritesNothingAgain() + { + MakeHistory1(); + var afterWrite = ReadPersistedLines(); + Assert.That( + afterWrite, + Has.Count.EqualTo(1), + "sanity check: the message should have been written normally" + ); + + // sut + _messageLog.Flush(); + + Assert.That( + ReadPersistedLines(), + Is.EqualTo(afterWrite), + "Flush must not write messages the file already has" + ); + } + + /// + /// Message writes give up quickly on a busy log file so as not to stall the UI thread, + /// and leave what they could not write for the next message to carry out. At shutdown + /// there is no next message, so disposing the collection that owns the log must flush it. + /// + [Test] + public void TeamCollectionDispose_MessageCouldNotBeWrittenEarlier_FlushesItToTheLogFile() + { + using (var collectionFolder = new TemporaryFolder("FlushOnDispose_Collection")) + using (var repoFolder = new TemporaryFolder("FlushOnDispose_Repo")) + { + var logPath = TeamCollectionManager.GetTcLogPathFromLcPath( + collectionFolder.FolderPath + ); + RobustFile.WriteAllText(logPath, ""); + var log = new TeamCollectionMessageLog(logPath); + var tc = new TestFolderTeamCollection( + new Mock().Object, + collectionFolder.FolderPath, + repoFolder.FolderPath, + log + ); + + using (new FileStream(logPath, FileMode.Open, FileAccess.Write, FileShare.None)) + { + log.WriteMessage( + MessageAndMilestoneType.History, + "TeamCollection.CheckoutMsg", + "{0} checked out the book {1}", + "joe@somewhere.org", + "Joe hunts pigs" + ); + } + Assert.That( + RobustFile.ReadAllText(logPath), + Is.Empty, + "setup problem: the message should not have reached the file while it was held" + ); + + // sut + tc.Dispose(); + + Assert.That( + File.ReadAllLines(logPath), + Is.EqualTo(log.Messages.Select(m => m.ToPersistedForm).ToArray()), + "disposing the collection should have flushed the outstanding message" + ); + } + } + + private List ReadPersistedLines() + { + return File.ReadAllLines(_logFile.Path) + .Where(line => !string.IsNullOrEmpty(line)) + .ToList(); + } + [Test] public void WriteMessage_RaisesStatusChanged() { @@ -590,6 +791,68 @@ public void WriteMessage_LogFileNotWritable_DoesNotThrowAndKeepsMessageInMemory( } } + /// + /// Repeating most errors is noise, so the log drops a duplicate whenever it occurred this + /// session. "Bloom may have missed some changes, please click Reload Collection" is the + /// exception: once the user has reloaded, another one is telling them something new, and + /// dropping it leaves them believing they are up to date. See BL-16729. + /// + [Test] + public void WriteMessage_MayHaveMissedChanges_SuppressedUntilAReload_ThenAllowedAgain() + { + void WarnOfMissedChanges() => + _messageLog.WriteMessage( + MessageAndMilestoneType.Error, + Bloom.TeamCollection.TeamCollection.kMayHaveMissedChangesId, + "Bloom may have missed some changes your teammates made.", + null, + null + ); + + WarnOfMissedChanges(); + Assert.That( + _messageLog.Messages, + Has.Count.EqualTo(1), + "setup problem: the first warning should always be recorded" + ); + + // Still the same unanswered warning, so the second one is genuinely redundant. + WarnOfMissedChanges(); + Assert.That( + _messageLog.Messages, + Has.Count.EqualTo(1), + "a repeat before the user has reloaded is redundant and should be dropped" + ); + + // The user reloads, which is exactly what the message asked for. + _messageLog.WriteMilestone(MessageAndMilestoneType.Reloaded); + + // sut: it happened again since. That is new information. + WarnOfMissedChanges(); + Assert.That( + _messageLog.Messages.Count(m => + m.L10NId == Bloom.TeamCollection.TeamCollection.kMayHaveMissedChangesId + ), + Is.EqualTo(2), + "after a reload the warning must be allowed through again, or the user is left " + + "believing they are up to date" + ); + + // And the narrowing is confined to that one message: ordinary errors still de-duplicate + // across the whole session, which is what keeps a repeating repo problem from flooding + // the log. + MakeError1(); + var countAfterFirst = _messageLog.Messages.Count; + _messageLog.WriteMilestone(MessageAndMilestoneType.Reloaded); + MakeError1(); + Assert.That( + _messageLog.Messages, + Has.Count.EqualTo(countAfterFirst + 1), + "only the Reloaded milestone should have been added; an ordinary error must still " + + "be treated as redundant even after a reload" + ); + } + [Test] public void WriteMessage_RedundantMessages_NotAdded() { diff --git a/src/BloomTests/TeamCollection/TestFolderTeamCollection.cs b/src/BloomTests/TeamCollection/TestFolderTeamCollection.cs index 2e4dae80f64c..39c2ed25e997 100644 --- a/src/BloomTests/TeamCollection/TestFolderTeamCollection.cs +++ b/src/BloomTests/TeamCollection/TestFolderTeamCollection.cs @@ -18,10 +18,69 @@ public TestFolderTeamCollection( ) : base(tcManager, localCollectionFolder, repoFolderPath, tcLog) { } + /// + /// HasCheckoutChangedRemotely is protected, and it is the whole of the new judgment the + /// catch-up scan makes, so let a test drive it directly rather than through the scan's + /// event plumbing. + /// + public bool CallHasCheckoutChangedRemotely(string bookName) => + HasCheckoutChangedRemotely(bookName); + + /// + /// Write the repo's copy of a book's status and nothing else. WriteBookStatus deliberately + /// writes the local copy too, which is right for something we did, but it is precisely + /// wrong for simulating what a teammate did on another machine: from here, their checkout + /// only ever shows up in the repo. + /// + public void WriteRepoStatusOnly(string bookName, BookStatus status) => + WriteBookStatusJsonToRepo(bookName, status.ToJson()); + public Action OnCreatedCalled; public Action OnChangedCalled; public Action OnCollectionChangedCalled; + /// + /// Lets a test pretend this is (or is not) the collection the manager is using, without + /// standing up a whole live TeamCollectionManager. Null means "use the real answer". + /// + public bool? PretendIsLiveCollection; + + protected internal override bool IsLiveCollection => + PretendIsLiveCollection ?? base.IsLiveCollection; + + /// + /// Lets a test pretend a repo write is in progress. + /// + public bool PretendIsWritingToRepo; + + protected internal override bool IsWritingToRepo => + PretendIsWritingToRepo || base.IsWritingToRepo; + + /// + /// Set InterceptCheckConnection to have CheckConnection report PretendConnectionProblem + /// instead of really looking, so a test can drive ConnectionHeartbeat.UpdateTcConnectionStatus with no + /// network and no real repo. Left off, the real implementation runs. + /// + public bool InterceptCheckConnection; + public TeamCollectionMessage PretendConnectionProblem; + public int CheckConnectionCallCount; + + /// + /// Set to have the intercepted CheckConnection throw instead of answering, so a test can + /// cover what the periodic check does when the probe itself fails. + /// + public bool PretendCheckConnectionThrows; + + public override TeamCollectionMessage CheckConnection(bool writeHistoryMessages) + { + CheckConnectionCallCount++; + if (!InterceptCheckConnection) + return base.CheckConnection(writeHistoryMessages); + if (PretendCheckConnectionThrows) + throw new IOException("pretend the probe itself blew up"); + return PretendConnectionProblem; + } + protected override void OnCreated(object sender, FileSystemEventArgs e) { base.OnCreated(sender, e);