diff --git a/.github/workflows/build-docs.sh b/.github/workflows/build-docs.sh deleted file mode 100755 index c526227af..000000000 --- a/.github/workflows/build-docs.sh +++ /dev/null @@ -1,39 +0,0 @@ -#!/bin/bash - -set -xe - -sudo apt-get update -sudo apt-get install -y python3-pip - -python3 -m venv ./venv -source venv/bin/activate - -pip3 install sphinx==8.1.3 -cd docs -make html - -cd build -git clone --depth 1 https://github.com/Shopify/ghostferry.git -b gh-pages ghostferry-pages -current_branch=${GITHUB_REF#refs/heads/} -cp -ar html/. ghostferry-pages/${current_branch} -cd ghostferry-pages - -echo "" > index.html -echo " " >> index.html -echo " Ghostferry Documentations" >> index.html -echo " " >> index.html -echo " " >> index.html -echo "

Ghostferry Documentation Version Selector

" >> index.html -echo " " >> index.html -echo " " >> index.html -echo "" >> index.html - -git status - -cd ../../.. -ls -la docs/build/ghostferry-pages -rm .git -rf diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index e62848649..e9b03bdfc 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -1,22 +1,70 @@ -name: Documentation on github pages +name: Documentation on: + pull_request: push: - branches: - - main + branches: [main] + +permissions: + contents: read + +concurrency: + group: docs-${{ github.ref }} + cancel-in-progress: true jobs: - github-pages: + build: runs-on: ubuntu-latest + env: + BUNDLE_WITHOUT: "development:test" + JEKYLL_ENV: production steps: - uses: actions/checkout@v6.0.2 + with: + persist-credentials: false + + - uses: ruby/setup-ruby@v1 + with: + bundler-cache: true + + - name: Build documentation + run: bundle exec jekyll build --source docs --destination build/docs + + - name: Check links and anchors + run: bundle exec htmlproofer build/docs --disable-external --allow-missing-href --no-enforce-https --swap-urls '^/ghostferry/main/:/' - - name: Build documentations - run: .github/workflows/build-docs.sh + - name: Upload documentation + uses: actions/upload-artifact@v7.0.1 + with: + name: docs-html + path: build/docs + if-no-files-found: error + retention-days: 7 + deploy: + runs-on: ubuntu-latest + needs: build + if: github.event_name == 'push' && github.ref == 'refs/heads/main' && github.repository == 'Shopify/ghostferry' + permissions: + contents: write + steps: + - uses: actions/checkout@v6.0.2 + with: + persist-credentials: false + + - name: Download documentation + uses: actions/download-artifact@v8.0.1 + with: + name: docs-html + path: build/docs - - name: Deploy github pages + - name: Deploy current documentation uses: peaceiris/actions-gh-pages@4f9cc6602d3f66b9c108549d475ec49e8ef4d45e # v4.0.0 with: github_token: ${{ secrets.GITHUB_TOKEN }} - publish_dir: ./docs/build/ghostferry-pages + publish_branch: gh-pages + publish_dir: ./build/docs + destination_dir: main + keep_files: false + force_orphan: false + enable_jekyll: false diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 39cde138d..b6a94e625 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -80,7 +80,7 @@ jobs: env: CI: "true" - BUNDLE_WITHOUT: "development" + BUNDLE_WITHOUT: "development:docs" MYSQL_VERSION: ${{ matrix.mysql }} GHOSTFERRY_LOG_BACKEND: ${{ matrix.log_backend }} diff --git a/Gemfile b/Gemfile index 2afad33c9..3a64d0706 100644 --- a/Gemfile +++ b/Gemfile @@ -17,4 +17,12 @@ group :development do gem "pry-byebug" end +group :docs do + gem "jekyll", "~> 4.4" + gem "just-the-docs", "~> 0.12" + gem "jekyll-relative-links", "~> 0.7" + gem "jekyll-optional-front-matter", "~> 0.3" + gem "html-proofer", "~> 5.0" +end + gem "mutex_m", "~> 0.3.0" diff --git a/Gemfile.lock b/Gemfile.lock index e4c140758..5ace8765d 100644 --- a/Gemfile.lock +++ b/Gemfile.lock @@ -1,14 +1,119 @@ GEM remote: https://rubygems.org/ specs: + Ascii85 (2.0.1) + addressable (2.9.0) + public_suffix (>= 2.0.2, < 8.0) + afm (1.0.0) ansi (1.6.0) + async (2.46.0) + console (~> 1.29) + fiber-annotation + io-event (~> 1.21) + base64 (0.3.0) + benchmark (0.5.0) bigdecimal (4.1.1) builder (3.3.0) byebug (13.0.0) reline (>= 0.6.0) coderay (1.1.3) + colorator (1.1.0) + concurrent-ruby (1.3.8) + console (1.38.0) + fiber-annotation + fiber-local (~> 1.1) + json + csv (3.3.6) + em-websocket (0.5.3) + eventmachine (>= 0.12.9) + http_parser.rb (~> 0) + ethon (0.18.0) + ffi (>= 1.15.0) + logger + eventmachine (1.2.7) + ffi (1.17.4) + ffi (1.17.4-arm64-darwin) + ffi (1.17.4-x86_64-linux-gnu) + fiber-annotation (0.2.0) + fiber-local (1.1.0) + fiber-storage + fiber-storage (1.0.1) + forwardable-extended (2.6.0) + google-protobuf (4.36.2) + bigdecimal + rake (~> 13.3) + google-protobuf (4.36.2-arm64-darwin) + bigdecimal + rake (~> 13.3) + google-protobuf (4.36.2-x86_64-linux-gnu) + bigdecimal + rake (~> 13.3) + hashery (2.1.2) + html-proofer (5.2.2) + addressable (~> 2.3) + async (~> 2.1) + benchmark (~> 0.5) + nokogiri (~> 1.13) + pdf-reader (~> 2.11) + rainbow (~> 3.0) + typhoeus (~> 1.3) + yell (~> 2.0) + zeitwerk (~> 2.5) + http_parser.rb (0.8.1) + i18n (1.15.2) + concurrent-ruby (~> 1.0) io-console (0.8.2) + io-event (1.22.1) + jekyll (4.4.1) + addressable (~> 2.4) + base64 (~> 0.2) + colorator (~> 1.0) + csv (~> 3.0) + em-websocket (~> 0.5) + i18n (~> 1.0) + jekyll-sass-converter (>= 2.0, < 4.0) + jekyll-watch (~> 2.0) + json (~> 2.6) + kramdown (~> 2.3, >= 2.3.1) + kramdown-parser-gfm (~> 1.0) + liquid (~> 4.0) + mercenary (~> 0.3, >= 0.3.6) + pathutil (~> 0.9) + rouge (>= 3.0, < 5.0) + safe_yaml (~> 1.0) + terminal-table (>= 1.8, < 4.0) + webrick (~> 1.7) + jekyll-include-cache (0.2.2) + jekyll (>= 3.7, < 5.0) + jekyll-optional-front-matter (0.3.3) + jekyll (>= 3.0, < 5.0) + jekyll-relative-links (0.8.0) + jekyll (>= 3.3, < 5.0) + jekyll-sass-converter (3.1.0) + sass-embedded (~> 1.75) + jekyll-seo-tag (2.9.0) + jekyll (>= 3.8, < 5.0) + jekyll-watch (2.2.1) + listen (~> 3.0) + json (2.21.2) + just-the-docs (0.12.0) + jekyll (>= 3.8.5) + jekyll-include-cache + jekyll-seo-tag (>= 2.0) + rake (>= 12.3.1) + kramdown (2.5.2) + rexml (>= 3.4.4) + kramdown-parser-gfm (1.1.0) + kramdown (~> 2.0) + liquid (4.0.4) + listen (3.10.0) + logger + rb-fsevent (~> 0.10, >= 0.10.3) + rb-inotify (~> 0.9, >= 0.9.10) + logger (1.7.0) + mercenary (0.4.0) method_source (1.1.0) + mini_portile2 (2.8.9) minitest (5.27.0) minitest-fail-fast (0.1.0) minitest (~> 5) @@ -24,6 +129,20 @@ GEM mutex_m (0.3.0) mysql2 (0.5.7) bigdecimal + nokogiri (1.19.4) + mini_portile2 (~> 2.8.2) + racc (~> 1.4) + nokogiri (1.19.4-arm64-darwin) + racc (~> 1.4) + nokogiri (1.19.4-x86_64-linux-gnu) + racc (~> 1.4) + pathutil (0.16.2) + forwardable-extended (~> 2.6) + pdf-reader (2.16.0) + Ascii85 (>= 1.0, < 3.0, != 2.0.0) + afm (>= 0.2.1, < 2) + hashery (~> 2.0) + ttfunk pry (0.16.0) coderay (~> 1.1) method_source (~> 1.0) @@ -31,17 +150,44 @@ GEM pry-byebug (3.12.0) byebug (~> 13.0) pry (>= 0.13, < 0.17) + public_suffix (7.0.5) + racc (1.8.1) + rainbow (3.1.1) rake (13.4.1) + rb-fsevent (0.11.2) + rb-inotify (0.11.1) + ffi (~> 1.0) reline (0.6.3) io-console (~> 0.5) + rexml (3.4.4) + rouge (4.7.0) ruby-progressbar (1.13.0) + safe_yaml (1.0.5) + sass-embedded (1.105.0) + google-protobuf (~> 4.31) + rake (~> 13.3) + terminal-table (3.0.2) + unicode-display_width (>= 1.1.1, < 3) tqdm (0.4.1) + ttfunk (1.7.0) + typhoeus (1.6.0) + ethon (>= 0.18.0) + unicode-display_width (2.6.0) webrick (1.9.2) + yell (2.2.2) + zeitwerk (2.8.3) PLATFORMS + arm64-darwin-23 ruby + x86_64-linux DEPENDENCIES + html-proofer (~> 5.0) + jekyll (~> 4.4) + jekyll-optional-front-matter (~> 0.3) + jekyll-relative-links (~> 0.7) + just-the-docs (~> 0.12) minitest minitest-fail-fast (~> 0.1.0) minitest-hooks diff --git a/README.md b/README.md index 15f0a8765..446ceb454 100644 --- a/README.md +++ b/README.md @@ -29,6 +29,29 @@ On a high-level, Ghostferry is broken into several components, enabling it to copy data. This is documented at https://shopify.github.io/ghostferry/main/technicaloverview.html +Documentation +------------- + +The documentation is written in Markdown under `docs/` and can be read +directly on GitHub, starting at the [Documentation source](docs/index.md). The +published site is built with [Jekyll](https://jekyllrb.com/) and the +[Just the Docs](https://just-the-docs.com/) theme; its settings, page titles +and navigation order live in `docs/_config.yml`. + +Internal contributors get the gems from `dev up`, then can run `dev docs` +(live preview) or `dev docs-build` (build plus link check). Otherwise: + +```bash +bundle install +bundle exec jekyll build --source docs --destination build/docs +bundle exec htmlproofer build/docs --disable-external --allow-missing-href --no-enforce-https --swap-urls '^/ghostferry/main/:/' +bundle exec jekyll serve --source docs --destination build/docs --host 127.0.0.1 --port 4000 +``` + +The build writes the site to `build/docs/`; `htmlproofer` fails on broken +internal links or anchors. The live preview is served at +http://127.0.0.1:4000/ghostferry/main/. None of these commands deploy anything. + Development Setup ----------------- diff --git a/dev.yml b/dev.yml index bcfe6b9c0..550bf1df8 100644 --- a/dev.yml +++ b/dev.yml @@ -31,3 +31,11 @@ commands: test-ruby: desc: Run the ruby test suite. run: make test-ruby + docs: + desc: Serve the documentation locally at http://127.0.0.1:4000/ghostferry/main/ (no deploy). + run: bundle exec jekyll serve --source docs --destination build/docs --host 127.0.0.1 --port 4000 + docs-build: + desc: Build the documentation into build/docs and check its links and anchors (no deploy). + run: | + bundle exec jekyll build --source docs --destination build/docs + bundle exec htmlproofer build/docs --disable-external --allow-missing-href --no-enforce-https --swap-urls '^/ghostferry/main/:/' diff --git a/docs/Makefile b/docs/Makefile deleted file mode 100644 index d2a4438d5..000000000 --- a/docs/Makefile +++ /dev/null @@ -1,20 +0,0 @@ -# Minimal makefile for Sphinx documentation -# - -# You can set these variables from the command line. -SPHINXOPTS = -SPHINXBUILD = python3 -msphinx -SPHINXPROJ = Ghostferry -SOURCEDIR = source -BUILDDIR = build - -# Put it first so that "make" without argument is like "make help". -help: - @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) - -.PHONY: help Makefile - -# Catch-all target: route all unknown targets to Sphinx using the new -# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS). -%: Makefile - @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) diff --git a/docs/_config.yml b/docs/_config.yml new file mode 100644 index 000000000..640a49c14 --- /dev/null +++ b/docs/_config.yml @@ -0,0 +1,72 @@ +title: Ghostferry +url: https://shopify.github.io +baseurl: /ghostferry/main +theme: just-the-docs +# Dark scheme, compiled into just-the-docs-default.css. The light scheme needs a +# matching assets/css/just-the-docs-.scss; _includes/head_custom.html +# picks between them (system setting or the reader's saved choice). +color_scheme: tokyonight-night +light_color_scheme: tokyonight-day +# Keeps Jekyll from writing docs/.jekyll-cache/ into the source tree. +disable_disk_cache: true + +markdown: kramdown +kramdown: + input: GFM + hard_wrap: false + syntax_highlighter: rouge + +plugins: + - jekyll-optional-front-matter + - jekyll-relative-links + +relative_links: + enabled: true + collections: false + +sass: + quiet_deps: true + silence_deprecations: [import, global-builtin, color-functions] + +optional_front_matter: + remove_originals: true + +include: + - _static +# Never publish a stale docs/build/ left behind by the old Sphinx build. +exclude: + - build/ + +search_enabled: true +heading_anchors: true +aux_links: + GitHub: https://github.com/Shopify/ghostferry +gh_edit_link: true +gh_edit_repository: https://github.com/Shopify/ghostferry +gh_edit_branch: main +gh_edit_source: docs +gh_edit_view_mode: edit + +# Page titles and order live here so the Markdown files need no front matter. +defaults: + - scope: { path: "" } + values: { layout: default } + # Generated theme assets (e.g. just-the-docs-head-nav.css) must not get the HTML layout. + - scope: { path: assets } + values: { layout: null } + - scope: { path: index.md } + values: { title: Home, nav_order: 1 } + - scope: { path: introduction.md } + values: { title: Introduction, nav_order: 2 } + - scope: { path: technicaloverview.md } + values: { title: Technical overview, nav_order: 3 } + - scope: { path: tutorialcopydb.md } + values: { title: Copydb tutorial, nav_order: 4 } + - scope: { path: copydbinprod.md } + values: { title: Running in production, nav_order: 5 } + - scope: { path: copydbinterruptresume.md } + values: { title: Interrupt and resume, nav_order: 6 } + - scope: { path: verifiers.md } + values: { title: Verifiers, nav_order: 7 } + - scope: { path: howtousecustom.md } + values: { title: Custom applications, nav_order: 8 } diff --git a/docs/_includes/head_custom.html b/docs/_includes/head_custom.html new file mode 100644 index 000000000..ec9c9429e --- /dev/null +++ b/docs/_includes/head_custom.html @@ -0,0 +1,55 @@ +{%- comment -%} + Light/dark theme selection. just-the-docs-default.css holds the dark scheme + (site.color_scheme); the light scheme (site.light_color_scheme) is written in + below. Media queries follow the system setting until the reader picks a theme + with the header button (saved in localStorage). + Keep this a blocking inline script: it runs before first paint, so a page + never renders in the wrong scheme. +{%- endcomment -%} +{%- capture light_css -%}{{ '/assets/css/just-the-docs-' | append: site.light_color_scheme | append: '.css' | relative_url }}{%- endcapture -%} + +{%- comment -%} Without JavaScript the dark sheet stays on; the light sheet overrides it when the system is light. {%- endcomment -%} + diff --git a/docs/_includes/header_custom.html b/docs/_includes/header_custom.html new file mode 100644 index 000000000..e458e8fe1 --- /dev/null +++ b/docs/_includes/header_custom.html @@ -0,0 +1,3 @@ +{%- comment -%} Cycles System → Light → Dark; wired up by head_custom.html. Hidden without JavaScript. {%- endcomment -%} + + diff --git a/docs/_sass/color_schemes/_tokyonight.scss b/docs/_sass/color_schemes/_tokyonight.scss new file mode 100644 index 000000000..98370564c --- /dev/null +++ b/docs/_sass/color_schemes/_tokyonight.scss @@ -0,0 +1,57 @@ +// Shared Tokyo Night mapping for Just the Docs. Import it from a variant file +// (tokyonight-night.scss / tokyonight-day.scss) that first defines the +// palette below, taken from https://github.com/folke/tokyonight.nvim, plus +// $color-scheme, $tn-text (body text) and $tn-button (primary buttons). + +$body-background-color: $tn-bg; +$body-heading-color: $tn-fg; +$body-text-color: $tn-text; +$link-color: $tn-blue; +$nav-child-link-color: $tn-dark5; +$sidebar-color: $tn-bg-dark; +$base-button-color: $tn-bg-highlight; +$btn-primary-color: $tn-button; +$code-background-color: $tn-bg-dark; +// Despite the name, Just the Docs uses this as the plain text colour of code blocks. +$code-linenumber-color: $tn-fg; +$feedback-color: $tn-bg-dark; +$table-background-color: $tn-bg-dark; +$search-background-color: $tn-bg-highlight; +$search-result-preview-color: $tn-fg-dark; +$border-color: $tn-terminal-black; + +// Syntax highlighting (Rouge emits Pygments class names). +.highlight { background: $tn-bg-dark; color: $tn-fg; } +.highlight .hll { background-color: $tn-bg-highlight; } +.highlight .c, .highlight .ch, .highlight .cm, .highlight .cpf, +.highlight .c1, .highlight .cs { color: $tn-comment; font-style: italic; } +.highlight .cp { color: $tn-cyan; font-style: normal; font-weight: normal; } +.highlight .err { color: $tn-red1; } +.highlight .esc, .highlight .g, .highlight .n, .highlight .x, +.highlight .nv, .highlight .vc, .highlight .vg, .highlight .vi, .highlight .vm, +.highlight .nx, .highlight .py { color: $tn-fg; } +.highlight .p { color: $tn-fg-dark; } +.highlight .o, .highlight .ow { color: $tn-blue5; font-weight: normal; } +.highlight .k, .highlight .kd, .highlight .kn, .highlight .kp, +.highlight .kr, .highlight .kv { color: $tn-magenta; font-weight: normal; } +.highlight .kc { color: $tn-orange; font-weight: normal; } +.highlight .kt, .highlight .nc, .highlight .nb, .highlight .bp, +.highlight .nn { color: $tn-blue1; } +.highlight .nf, .highlight .fm, .highlight .na, .highlight .nt { color: $tn-blue; } +.highlight .no, .highlight .ne, .highlight .nl, .highlight .nd, +.highlight .ni { color: $tn-orange; } +.highlight .l, .highlight .ld, .highlight .s, .highlight .sa, .highlight .sb, +.highlight .sc, .highlight .dl, .highlight .sd, .highlight .s2, .highlight .sh, +.highlight .s1, .highlight .ss { color: $tn-green; } +.highlight .se, .highlight .si, .highlight .sr, .highlight .sx { color: $tn-blue5; } +.highlight .m, .highlight .mb, .highlight .mf, .highlight .mh, .highlight .mi, +.highlight .il, .highlight .mo { color: $tn-orange; } +.highlight .gp { color: $tn-magenta; font-weight: normal; } +.highlight .go { color: $tn-fg-dark; } +.highlight .gh, .highlight .gu { color: $tn-blue; font-weight: bold; } +.highlight .gd { color: $tn-red; background-color: transparent; } +.highlight .gi { color: $tn-green1; background-color: transparent; } +.highlight .gr, .highlight .gt { color: $tn-red1; } +.highlight .ge { font-style: italic; } +.highlight .gs { font-weight: bold; } +.highlight .w { color: $tn-fg-gutter; } diff --git a/docs/_sass/color_schemes/tokyonight-day.scss b/docs/_sass/color_schemes/tokyonight-day.scss new file mode 100644 index 000000000..500b3623a --- /dev/null +++ b/docs/_sass/color_schemes/tokyonight-day.scss @@ -0,0 +1,30 @@ +// Tokyo Night, "day" variant. +$tn-bg: #e1e2e7; +$tn-bg-dark: #d0d5e3; +$tn-bg-highlight: #c4c8da; +$tn-terminal-black: #a1a6c5; +$tn-fg: #3760bf; +$tn-fg-dark: #6172b0; +$tn-fg-gutter: #a8aecb; +$tn-comment: #848cb5; +$tn-dark5: #68709a; +$tn-blue0: #7890dd; +$tn-blue: #2e7de9; +$tn-cyan: #007197; +$tn-blue1: #188092; +$tn-blue5: #006a83; +$tn-magenta: #9854f1; +$tn-orange: #b15c00; +$tn-yellow: #8c6c3e; +$tn-green: #587539; +$tn-green1: #387068; +$tn-red: #f52a65; +$tn-red1: #c64343; + +$color-scheme: light; +// Day's fg_dark is too faint for long-form body text on its background. +$tn-text: $tn-fg; +// White button text needs the stronger blue on a light background. +$tn-button: $tn-blue; + +@import "./tokyonight"; diff --git a/docs/_sass/color_schemes/tokyonight-night.scss b/docs/_sass/color_schemes/tokyonight-night.scss new file mode 100644 index 000000000..850cacec4 --- /dev/null +++ b/docs/_sass/color_schemes/tokyonight-night.scss @@ -0,0 +1,28 @@ +// Tokyo Night, "night" variant. +$tn-bg: #1a1b26; +$tn-bg-dark: #16161e; +$tn-bg-highlight: #292e42; +$tn-terminal-black: #414868; +$tn-fg: #c0caf5; +$tn-fg-dark: #a9b1d6; +$tn-fg-gutter: #3b4261; +$tn-comment: #565f89; +$tn-dark5: #737aa2; +$tn-blue0: #3d59a1; +$tn-blue: #7aa2f7; +$tn-cyan: #7dcfff; +$tn-blue1: #2ac3de; +$tn-blue5: #89ddff; +$tn-magenta: #bb9af7; +$tn-orange: #ff9e64; +$tn-yellow: #e0af68; +$tn-green: #9ece6a; +$tn-green1: #73daca; +$tn-red: #f7768e; +$tn-red1: #db4b4b; + +$color-scheme: dark; +$tn-text: $tn-fg-dark; +$tn-button: $tn-blue0; + +@import "./tokyonight"; diff --git a/docs/_sass/custom/custom.scss b/docs/_sass/custom/custom.scss new file mode 100644 index 000000000..f21207628 --- /dev/null +++ b/docs/_sass/custom/custom.scss @@ -0,0 +1,8 @@ +// Theme toggle (_includes/header_custom.html): match the header's aux links. +.theme-toggle { + @include fs-2; + margin-left: auto; + color: $link-color; + white-space: nowrap; + cursor: pointer; +} diff --git a/docs/source/_static/ghostferry-architecture.png b/docs/_static/ghostferry-architecture.png similarity index 100% rename from docs/source/_static/ghostferry-architecture.png rename to docs/_static/ghostferry-architecture.png diff --git a/docs/source/_static/percona-talk.pdf b/docs/_static/percona-talk.pdf similarity index 100% rename from docs/source/_static/percona-talk.pdf rename to docs/_static/percona-talk.pdf diff --git a/docs/assets/css/just-the-docs-tokyonight-day.scss b/docs/assets/css/just-the-docs-tokyonight-day.scss new file mode 100644 index 000000000..1f82f15ea --- /dev/null +++ b/docs/assets/css/just-the-docs-tokyonight-day.scss @@ -0,0 +1,14 @@ +--- +--- +{% include css/just-the-docs.scss.liquid color_scheme="tokyonight-day" %} + +// just-the-docs-head-nav.css bakes this gradient with the default (Night) +// colours; repeat it here so Day wins while this sheet is active. +.site-nav ul li a { + background-image: linear-gradient( + -90deg, + rgba($feedback-color, 1) 0%, + rgba($feedback-color, 0.8) 80%, + rgba($feedback-color, 0) 100% + ); +} diff --git a/docs/source/copydbinprod.rst b/docs/copydbinprod.md similarity index 63% rename from docs/source/copydbinprod.rst rename to docs/copydbinprod.md index d9be4ca5b..5afe05110 100644 --- a/docs/source/copydbinprod.rst +++ b/docs/copydbinprod.md @@ -1,16 +1,13 @@ -.. _copydbinprod: + -=========================================== -Running ``ghostferry-copydb`` in production -=========================================== +# Running `ghostferry-copydb` in production -Assuming you have gone through :ref:`tutorialcopydb`, you probably want to run -``ghostferry-copydb`` in production. The general workflow is relatively +Assuming you have gone through [Tutorial for ghostferry-copydb](tutorialcopydb.md), you probably want to run +`ghostferry-copydb` in production. The general workflow is relatively similar, with some differences. You should keep the tutorial as a starting point for your own playbook as most steps will largely be the same. -Prerequisites -------------- +## Prerequisites Before you start, you need to know if you can even use Ghostferry. Some points to consider about this are: @@ -18,43 +15,39 @@ to consider about this are: - Ghostferry on its own does not enable zero downtime moves. The downtime for the app will be minimal compared to other methods but still non-zero. - - Figure out how much downtime you are willing to tolerate. Using Ghostferry, - one can realistically achieve downtime in the order of seconds. + - Figure out how much downtime you are willing to tolerate. Using Ghostferry, + one can realistically achieve downtime in the order of seconds. -- The source database must be running with `FULL image`_ `ROW based replication`_. +- The source database must be running with [FULL image](https://dev.mysql.com/doc/refman/5.7/en/replication-options-binary-log.html#sysvar_binlog_row_image) [ROW based replication](https://dev.mysql.com/doc/refman/5.6/en/replication-options-binary-log.html#sysvar_binlog_format). - - Without this, it is not possible to run Ghostferry safely and Ghostferry - will error out if it detects ``binlog_row_image`` is not set to ``FULL``. + - Without this, it is not possible to run Ghostferry safely and Ghostferry + will error out if it detects `binlog_row_image` is not set to `FULL`. - Tables to be copied have integer primary keys. - - An issue exists to fix this limitation here: - ``_. - - As a work around, you can use mysqldump during the cutover stage to migrate - those tables. + - An issue exists to fix this limitation here: + . + - As a work around, you can use mysqldump during the cutover stage to migrate + those tables. - There are no foreign key constraints in your tables. - - You should remove these constraints before running Ghostferry. + - You should remove these constraints before running Ghostferry. -- ``ghostferry-copydb`` can only copy a whole table at a time. +- `ghostferry-copydb` can only copy a whole table at a time. - - If you need to copy a subset, use ghostferry as a library to build your own - application. + - If you need to copy a subset, use ghostferry as a library to build your own + application. - In cases of a multi-node replication setup, Ghostferry should only be run on the master database where writes occur. Otherwise there may be a race condition causing some binlog entries to be missed. - - There may be a way to fix this in the future. + - There may be a way to fix this in the future. -.. _`FULL image`: https://dev.mysql.com/doc/refman/5.7/en/replication-options-binary-log.html#sysvar_binlog_row_image -.. _`ROW based replication`: https://dev.mysql.com/doc/refman/5.6/en/replication-options-binary-log.html#sysvar_binlog_format + -.. _prodtesting: - -Testing Ghostferry with Production Data ---------------------------------------- +## Testing Ghostferry with Production Data You can run Ghostferry without running the cutover to test the entire flow without actually moving your database. This allows you to verify that the move @@ -72,8 +65,7 @@ some additional setup: database to read only. 4. Perform the cutover as normal. Can even run verification with this. -To Verify Or Not To Verify --------------------------- +## To Verify Or Not To Verify Ghostferry has two built-in verifiers. They are designed to give you certainty that the data of the source and the target are identical after a move and @@ -81,13 +73,13 @@ nothing was corrupted/missed. They are designed to be used during the cutover process, when writes to the source database have stopped and writes to the target have not yet started. During the run, both the source and target must be kept read only and thus incur downtime for your dataset. The two different -verifiers have different downtime characteristics. See the :ref:`verifiers` +verifiers have different downtime characteristics. See the [Verifiers](verifiers.md) page for more details on what they are and how to choose a verifier. This means you have to decide if you want to verify or not. In order to know how much downtime you will incur during the verification process, you can test it with the slave based staging move described in -`Testing Ghostferry with Production Data`_. During the cutover stages, run +[Testing Ghostferry with Production Data](#prodtesting). During the cutover stages, run verification as normal and measure the time taken. Since the ultimate objective of the verifier is to verify that Ghostferry did @@ -106,36 +98,31 @@ issues after the fact: 3. Run ghostferry as normal between the master source and target database. 4. During the cutover, also stop replication to the slaves setup in step 1 and 2. -5. Manually compare the table on the slaves using something like ``CHECKSUM - TABLE``. +5. Manually compare the table on the slaves using something like `CHECKSUM TABLE`. -Dealing with Errors and Restarting Runs ---------------------------------------- +## Dealing with Errors and Restarting Runs It is possible for Ghostferry to encounter an unrecoverable error (such as a network partition with the database). In these scenarios, the target will be left alone as the Ghostferry process panics and quits. It may be possible to resume these runs using the experimental interrupt & resume feature. See -:ref:`copydbinterruptresume`. +[Interrupt and resuming `ghostferry-copydb`](copydbinterruptresume.md). If the resume doesn't work, starting a brand new Ghostferry run is perfectly fine. For copydb specifically, you need to drop the databases created by copydb on the target as it will try to recreate it. -Configuration for ``ghostferry-copydb`` ---------------------------------------- +## Configuration for `ghostferry-copydb` -The configuration for ``ghostferry-copydb`` is a JSON file. The schema it is -based on the `Config struct of ghostferry -`__, with some +The configuration for `ghostferry-copydb` is a JSON file. The schema it is +based on the [Config struct of ghostferry](https://godoc.org/github.com/Shopify/ghostferry#Config), with some differences: -- You cannot specify ``TableFilter`` and ``CopyFilter``. +- You cannot specify `TableFilter` and `CopyFilter`. - If you are using the debian package, you don't need to specify - ``WebBasedir`` as it is compiled into the binary. + `WebBasedir` as it is compiled into the binary. It also allows you specify some options according to fields defined by the -`Config struct of copydb -`__. This allows +[Config struct of copydb](https://godoc.org/github.com/Shopify/ghostferry/copydb#Config). This allows you to filter the databases/tables to copy as well as specify the type of verifier. diff --git a/docs/copydbinterruptresume.md b/docs/copydbinterruptresume.md new file mode 100644 index 000000000..103cf9ff4 --- /dev/null +++ b/docs/copydbinterruptresume.md @@ -0,0 +1,115 @@ + + +# Interrupt and resuming `ghostferry-copydb` + +*Note that this is a new and experimental feature. Please ensure you test it +thoroughly in your environment to ensure there are no data loss. See the bottom +of this page for important information on caveats of using this feature.* + +To enable state dumps of Ghostferry on panic (and thus interrupt & resume), the +configuration given to copydb must have the entry `"DumpStateOnSignal": true`. +Once this is configured, any time the process panics, which can be +caused by both SIGTERM/SIGINT or due to an error within Ghostferry, the run +state will be dumped to stdout with JSON. An example of this can be seen below: + +```json +{ + "GhostferryVersion": "1.1.0+20190311205252+88a1c5c", + "LastKnownTableSchemaCache": { + "abc.table1": { + "Schema": "abc", + "Name": "table1", + "Columns": [ + { + "Name": "id", + "Type": 1, + "Collation": "", + "RawType": "bigint(20)", + "IsAuto": true, + "IsUnsigned": false, + "EnumValues": null, + "SetValues": null + }, + { + "Name": "data", + "Type": 5, + "Collation": "utf8mb4_unicode_ci", + "RawType": "varchar(16)", + "IsAuto": false, + "IsUnsigned": false, + "EnumValues": null, + "SetValues": null + } + ], + "Indexes": [ + { + "Name": "PRIMARY", + "Columns": [ + "id" + ], + "Cardinality": [ + 1 + ] + } + ], + "PKColumns": [ + 0 + ], + "UnsignedColumns": null + } + }, + "CurrentStage": "COPY", + "CopyStage": { + "LastProcessedBinlogPosition": { + "Name": "mysql-bin.000008", + "Pos": 193989 + }, + "LastSuccessfulPaginationKeys": { + "abc.table1": 200 + }, + "CompletedTables": {} + }, + "VerifierStage": null +} +``` + +To resume, you first need to save this JSON into a file. Alternatively, you +could pipe the stdout of `ghostferry-copydb` directly into a file via: + +```console +$ ghostferry-copydb -verbose conf.json >state-dump.json 2>ghostferry.log +``` + +Theoretically, `ghostferry-copydb` should write only the state dump json into +stdout and all logs in stderr. However, check the files to make sure this is +true. If not, file a bug report. + +To resume, pass `state-dump.json` as a flag back to `ghostferry-copydb`: + +```console +$ ghostferry-copydb -verbose -resumestate state-dump.json conf.json +``` + +**Note: if you interrupt Ghostferry for a period of time longer than your +binlog retention time, you will not be able to resume Ghostferry. Ensure that +the binlog at the position recorded in the state dump is available when +resuming Ghostferry.** + +Some other considerations/notes: + +- While Ghostferry will dump the state when it encounters an unrecoverable + error (such as a network issue to the databases), the only tested use case + for now is due to an interrupt with SIGTERM/SIGINT. + + - Errored runs *should* be theoretically safe to resume, but this is not + validated in any form. If you resume an errored run, it is recommended to + validate the correctness of the data using the CHECKSUM TABLE verifier. + - As the project develops, we want to validate the safety of resuming errored + runs. + - To test resuming errored runs further, see + [Testing Ghostferry with Production Data](copydbinprod.md#prodtesting). + +- Verifiers are not resumable, including the IterativeVerifier. This may change + in the future. +- While we are confident the algorithm is correct, this is still a + highly experimental feature. USE AT YOUR OWN RISK. diff --git a/docs/howtousecustom.md b/docs/howtousecustom.md new file mode 100644 index 000000000..09ed36077 --- /dev/null +++ b/docs/howtousecustom.md @@ -0,0 +1,35 @@ + + +# Using Ghostferry in Custom Applications + +For an example application, see [ghostferry-copydb](https://github.com/Shopify/ghostferry/tree/main/copydb). + +## Consuming Ghostferry Metrics + +Ghostferry provides optional metrics to your application. + +Start consuming the metrics: + +```go +sink := make(chan interface{}, 512) +metrics := ghostferry.SetGlobalMetrics("myApp", sink) + +go func(){ + for { + switch metric := (<-sink).(type) { + case ghostferry.CountMetric: + // Do something with the metric + case ghostferry.GaugeMetric: + // Do something with the metric + case ghostferry.TimerMetric: + // Do something with the metric + } + } +}() +``` + +Emit additional metrics: + +```go +metrics.Count("myOwnCustomMetrics", 42, nil, 1.0) +``` diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 000000000..7bc335f91 --- /dev/null +++ b/docs/index.md @@ -0,0 +1,41 @@ + + +# Welcome to Ghostferry's documentation! + +Contents: + +- [Introduction to Ghostferry](introduction.md) + - [Why do I need this?](introduction.md#why-do-i-need-this) +- [Technical Overview](technicaloverview.md) + - [Architecture](technicaloverview.md#architecture) + - [Limitations](technicaloverview.md#limitations) + - [Algorithm Correctness](technicaloverview.md#algorithm-correctness) +- [Tutorial for ghostferry-copydb](tutorialcopydb.md) + - [Setup and Seed MySQL](tutorialcopydb.md#setup-and-seed-mysql) + - [(Mirrors Production) Create Ghostferry Users](tutorialcopydb.md#mirrors-production-create-ghostferry-users) + - [(Mirrors Production) Install ghostferry-copydb](tutorialcopydb.md#mirrors-production-install-ghostferry-copydb) + - [(Mirrors Production) Setup Ghostferry Run Configuration](tutorialcopydb.md#mirrors-production-setup-ghostferry-run-configuration) + - [(Mirrors Production) Validate Ghostferry Configuration](tutorialcopydb.md#mirrors-production-validate-ghostferry-configuration) + - [(Mirrors Production) Starting Ghostferry Run](tutorialcopydb.md#mirrors-production-starting-ghostferry-run) + - [(Mirrors Production) Monitoring Ghostferry Run via Web UI](tutorialcopydb.md#mirrors-production-monitoring-ghostferry-run-via-web-ui) + - [(Mirrors Production) Perform Cutover](tutorialcopydb.md#mirrors-production-perform-cutover) + - [(Mirrors Production) Verify Source and Target Data are Identical](tutorialcopydb.md#mirrors-production-verify-source-and-target-data-are-identical) + - [Finishing Ghostferry Run and Next Steps](tutorialcopydb.md#finishing-ghostferry-run-and-next-steps) +- [Running `ghostferry-copydb` in production](copydbinprod.md) + - [Prerequisites](copydbinprod.md#prerequisites) + - [Testing Ghostferry with Production Data](copydbinprod.md#testing-ghostferry-with-production-data) + - [To Verify Or Not To Verify](copydbinprod.md#to-verify-or-not-to-verify) + - [Dealing with Errors and Restarting Runs](copydbinprod.md#dealing-with-errors-and-restarting-runs) + - [Configuration for `ghostferry-copydb`](copydbinprod.md#configuration-for-ghostferry-copydb) +- [Interrupt and resuming `ghostferry-copydb`](copydbinterruptresume.md) +- [Verifiers](verifiers.md) + - [IterativeVerifier (Deprecated)](verifiers.md#iterativeverifier-deprecated) + - [InlineVerifier](verifiers.md#inlineverifier) + - [TargetVerifier](verifiers.md#targetverifier) +- [Using Ghostferry in Custom Applications](howtousecustom.md) + - [Consuming Ghostferry Metrics](howtousecustom.md#consuming-ghostferry-metrics) + +## Other resources + +- [API Documentations](https://godoc.org/github.com/Shopify/ghostferry) +- [**Percona Live Conference Slides + Presenter Notes**](_static/percona-talk.pdf) diff --git a/docs/source/introduction.rst b/docs/introduction.md similarity index 84% rename from docs/source/introduction.rst rename to docs/introduction.md index 8d3a44932..0ff5bfb88 100644 --- a/docs/source/introduction.rst +++ b/docs/introduction.md @@ -1,8 +1,6 @@ -.. _introduction: + -========================== -Introduction to Ghostferry -========================== +# Introduction to Ghostferry Ghostferry is a library that enables you to copy data from one MySQL instance to another with minimal amount of downtime. This is accomplished by tailing @@ -13,16 +11,15 @@ is because Ghostferry has the capability to selectively filter data to copy. The filtering could be arbitrarily complex and thus cannot be easily expressed in some configuration file. -That said, there is a generic tool called ``ghostferry-copydb`` that will copy +That said, there is a generic tool called `ghostferry-copydb` that will copy tables and the data contained in them from one MySQL to another with only the basic database/table name filtering. -Ghostferry is inspired by Github's `gh-ost `_. +Ghostferry is inspired by Github's [gh-ost](https://github.com/github/gh-ost). However, instead of copying data from and to the same database, Ghostferry copies data from one database to another. -Why do I need this? -=================== +## Why do I need this? Traditionally, moving data from one database to another involves some sort of backup and restore along with replaying the changes via replication. Backup is diff --git a/docs/source/conf.py b/docs/source/conf.py deleted file mode 100644 index b8a415aac..000000000 --- a/docs/source/conf.py +++ /dev/null @@ -1,177 +0,0 @@ -# -*- coding: utf-8 -*- -# -# Ghostferry documentation build configuration file, created by -# sphinx-quickstart on Tue Aug 15 10:55:18 2017. -# -# This file is execfile()d with the current directory set to its -# containing dir. -# -# Note that not all possible configuration values are present in this -# autogenerated file. -# -# All configuration values have a default; values that are commented out -# serve to show the default. - -# If extensions (or modules to document with autodoc) are in another directory, -# add these directories to sys.path here. If the directory is relative to the -# documentation root, use os.path.abspath to make it absolute, like shown here. -# -# import os -# import sys -# sys.path.insert(0, os.path.abspath('.')) - - -# -- General configuration ------------------------------------------------ - -# If your documentation needs a minimal Sphinx version, state it here. -# -# needs_sphinx = '1.0' - -# Add any Sphinx extension module names here, as strings. They can be -# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom -# ones. -extensions = ['sphinx.ext.githubpages'] - -# Add any paths that contain templates here, relative to this directory. -templates_path = ['_templates'] - -# The suffix(es) of source filenames. -# You can specify multiple suffix as a list of string: -# -# source_suffix = ['.rst', '.md'] -source_suffix = '.rst' - -# The master toctree document. -master_doc = 'index' - -# General information about the project. -project = u'Ghostferry' -copyright = u'2017-2024, Shopify' -author = u'Shopify' - -# The version info for the project you're documenting, acts as replacement for -# |version| and |release|, also used in various other places throughout the -# built documents. -# -# The short X.Y version. -version = u'1.0' -# The full version, including alpha/beta/rc tags. -release = u'1.0.0' - -# The language for content autogenerated by Sphinx. Refer to documentation -# for a list of supported languages. -# -# This is also used if you do content translation via gettext catalogs. -# Usually you set "language" from the command line for these cases. -language = 'en' - -# List of patterns, relative to source directory, that match files and -# directories to ignore when looking for source files. -# This patterns also effect to html_static_path and html_extra_path -exclude_patterns = [] - -# The name of the Pygments (syntax highlighting) style to use. -pygments_style = 'sphinx' - -# If true, `todo` and `todoList` produce output, else they produce nothing. -todo_include_todos = False - - -# -- Options for HTML output ---------------------------------------------- - -# The theme to use for HTML and HTML Help pages. See the documentation for -# a list of builtin themes. -# -html_theme = 'alabaster' - -# Theme options are theme-specific and customize the look and feel of a theme -# further. For a list of options available for each theme, see the -# documentation. -# -html_theme_options = { - "show_related": True, - "github_user": "Shopify", - "github_repo": "ghostferry", - "github_type": "star", - "description": " The swiss army knife of live data migrations", -} - -# Add any paths that contain custom static files (such as style sheets) here, -# relative to this directory. They are copied after the builtin static files, -# so a file named "default.css" will overwrite the builtin "default.css". -html_static_path = ['_static'] - -# Custom sidebar templates, must be a dictionary that maps document names -# to template names. -# -# This is required for the alabaster theme -# refs: http://alabaster.readthedocs.io/en/latest/installation.html#sidebars -html_sidebars = { - '**': [ - 'about.html', - 'navigation.html', - 'relations.html', # needs 'show_related': True theme option to display - 'searchbox.html', - 'donate.html', - ] -} - - -# -- Options for HTMLHelp output ------------------------------------------ - -# Output file base name for HTML help builder. -htmlhelp_basename = 'Ghostferrydoc' - - -# -- Options for LaTeX output --------------------------------------------- - -latex_elements = { - # The paper size ('letterpaper' or 'a4paper'). - # - # 'papersize': 'letterpaper', - - # The font size ('10pt', '11pt' or '12pt'). - # - # 'pointsize': '10pt', - - # Additional stuff for the LaTeX preamble. - # - # 'preamble': '', - - # Latex figure (float) alignment - # - # 'figure_align': 'htbp', -} - -# Grouping the document tree into LaTeX files. List of tuples -# (source start file, target name, title, -# author, documentclass [howto, manual, or own class]). -latex_documents = [ - (master_doc, 'Ghostferry.tex', u'Ghostferry Documentation', - u'Shopify', 'manual'), -] - - -# -- Options for manual page output --------------------------------------- - -# One entry per manual page. List of tuples -# (source start file, name, description, authors, manual section). -man_pages = [ - (master_doc, 'ghostferry', u'Ghostferry Documentation', - [author], 1) -] - - -# -- Options for Texinfo output ------------------------------------------- - -# Grouping the document tree into Texinfo files. List of tuples -# (source start file, target name, title, author, -# dir menu entry, description, category) -texinfo_documents = [ - (master_doc, 'Ghostferry', u'Ghostferry Documentation', - author, 'Ghostferry', 'One line description of project.', - 'Miscellaneous'), -] - - - diff --git a/docs/source/copydbinterruptresume.rst b/docs/source/copydbinterruptresume.rst deleted file mode 100644 index 5599a8b24..000000000 --- a/docs/source/copydbinterruptresume.rst +++ /dev/null @@ -1,117 +0,0 @@ -.. _copydbinterruptresume: - -============================================ -Interrupt and resuming ``ghostferry-copydb`` -============================================ - -*Note that this is a new and experimental feature. Please ensure you test it -thoroughly in your environment to ensure there are no data loss. See the bottom -of this page for important information on caveats of using this feature.* - -To enable state dumps of Ghostferry on panic (and thus interrupt & resume), the -configuration given to copydb must have the entry ``"DumpStateOnSignal": -true``. Once this is configured, any time the process panics, which can be -caused by both SIGTERM/SIGINT or due to an error within Ghostferry, the run -state will be dumped to stdout with JSON. An example of this can be seen below: - -.. code-block:: json - - { - "GhostferryVersion": "1.1.0+20190311205252+88a1c5c", - "LastKnownTableSchemaCache": { - "abc.table1": { - "Schema": "abc", - "Name": "table1", - "Columns": [ - { - "Name": "id", - "Type": 1, - "Collation": "", - "RawType": "bigint(20)", - "IsAuto": true, - "IsUnsigned": false, - "EnumValues": null, - "SetValues": null - }, - { - "Name": "data", - "Type": 5, - "Collation": "utf8mb4_unicode_ci", - "RawType": "varchar(16)", - "IsAuto": false, - "IsUnsigned": false, - "EnumValues": null, - "SetValues": null - } - ], - "Indexes": [ - { - "Name": "PRIMARY", - "Columns": [ - "id" - ], - "Cardinality": [ - 1 - ] - } - ], - "PKColumns": [ - 0 - ], - "UnsignedColumns": null - } - }, - "CurrentStage": "COPY", - "CopyStage": { - "LastProcessedBinlogPosition": { - "Name": "mysql-bin.000008", - "Pos": 193989 - }, - "LastSuccessfulPaginationKeys": { - "abc.table1": 200 - }, - "CompletedTables": {} - }, - "VerifierStage": null - } - - -To resume, you first need to save this JSON into a file. Alternatively, you -could pipe the stdout of ``ghostferry-copydb`` directly into a file via: - -.. code-block:: shell-session - - $ ghostferry-copydb -verbose conf.json >state-dump.json 2>ghostferry.log - -Theoretically, ``ghostferry-copydb`` should write only the state dump json into -stdout and all logs in stderr. However, check the files to make sure this is -true. If not, file a bug report. - -To resume, pass ``state-dump.json`` as a flag back to ``ghostferry-copydb``: - -.. code-block:: shell-session - - $ ghostferry-copydb -verbose -resumestate state-dump.json conf.json - -**Note: if you interrupt Ghostferry for a period of time longer than your -binlog retention time, you will not be able to resume Ghostferry. Ensure that -the binlog at the position recorded in the state dump is available when -resuming Ghostferry.** - -Some other considerations/notes: - -* While Ghostferry will dump the state when it encounters an unrecoverable - error (such as a network issue to the databases), the only tested use case - for now is due to an interrupt with SIGTERM/SIGINT. - - * Errored runs *should* be theoretically safe to resume, but this is not - validated in any form. If you resume an errored run, it is recommended to - validate the correctness of the data using the CHECKSUM TABLE verifier. - * As the project develops, we want to validate the safety of resuming errored - runs. - * To test resuming errored runs further, see :ref:`prodtesting`. - -* Verifiers are not resumable, including the IterativeVerifier. This may change - in the future. -* While we are confident the algorithm is correct, this is still a - highly experimental feature. USE AT YOUR OWN RISK. diff --git a/docs/source/howtousecustom.rst b/docs/source/howtousecustom.rst deleted file mode 100644 index ec684a304..000000000 --- a/docs/source/howtousecustom.rst +++ /dev/null @@ -1,34 +0,0 @@ -.. _howtousecustom: - -======================================= -Using Ghostferry in Custom Applications -======================================= - -TODO. For the time being, see the ghostferry-copydb project. - -Consuming Ghostferry Metrics ----------------------------- - -Ghostferry provides optional metrics to your application. - -Start consuming the metrics:: - - sink := make(chan interface{}, 512) - metrics := ghostferry.SetGlobalMetrics("myApp", sink) - - go func(){ - for { - switch metric := (<-sink).(type) { - case ghostferry.CountMetric: - // Do something with the metric - case ghostferry.GaugeMetric: - // Do something with the metric - case ghostferry.TimerMetric: - // Do something with the metric - } - } - }() - -Emit additional metrics:: - - metrics.Count("myOwnCustomMetrics", 42, nil, 1.0) diff --git a/docs/source/index.rst b/docs/source/index.rst deleted file mode 100644 index 1cba36880..000000000 --- a/docs/source/index.rst +++ /dev/null @@ -1,32 +0,0 @@ -.. Ghostferry documentation master file, created by - sphinx-quickstart on Tue Aug 15 10:55:18 2017. - You can adapt this file completely to your liking, but it should at least - contain the root `toctree` directive. - -Welcome to Ghostferry's documentation! -====================================== - -.. toctree:: - :maxdepth: 2 - :caption: Contents: - - introduction - technicaloverview - tutorialcopydb - copydbinprod - copydbinterruptresume - verifiers - howtousecustom - -Other resources -=============== - -- `API Documentations `_ -- `**Percona Live Conference Slides + Presenter Notes** <_static/percona-talk.pdf>`_ - -Indices and tables -================== - -* :ref:`genindex` -* :ref:`modindex` -* :ref:`search` diff --git a/docs/source/verifiers.rst b/docs/source/verifiers.rst deleted file mode 100644 index 8957e6921..000000000 --- a/docs/source/verifiers.rst +++ /dev/null @@ -1,200 +0,0 @@ -.. _verifiers: - -========= -Verifiers -========= - -Verifiers in Ghostferry are designed to ensure that Ghostferry did not -corrupt/miss data. There are three different verifiers: the -``ChecksumTableVerifier``, the ``InlineVerifier``, and the ``TargetVerifier``. A comparison of the -``ChecksumTableVerifier`` and ``InlineVerifier`` are given below: - -+-----------------------+-----------------------+-----------------------------+ -| | ChecksumTableVerifier | InlineVerifier | -+-----------------------+-----------------------+-----------------------------+ -|Mechanism | ``CHECKSUM TABLE`` | Verify row after insert; | -| | | Reverify changed rows before| -| | | and during cutover. | -+-----------------------+-----------------------+-----------------------------+ -|Impacts on Cutover Time| Linear w.r.t data size| Linear w.r.t. change rate | -| | | [1]_ | -+-----------------------+-----------------------+-----------------------------+ -|Impacts on Copy Time | None | Linear w.r.t data size | -|[2]_ | | | -+-----------------------+-----------------------+-----------------------------+ -|Memory Usage | Minimal | Linear w.r.t rows changed | -+-----------------------+-----------------------+-----------------------------+ -|Partial table copy | Not supported | Supported | -+-----------------------+-----------------------+-----------------------------+ -|Worst Case Scenario | Large databases causes| Verification is slower than | -| | unacceptable downtime | the change rate of the DB | -+-----------------------+-----------------------+-----------------------------+ - -.. [1] Additional improvements could be made to reduce this as long as - Ghostferry is faster than the rate of change. See - ``_. - -.. [2] Increase in copy time does not increase downtime. Downtime occurs only - in cutover. - -If you want verification, you should try with the ``ChecksumTableVerifier`` -first if you're copying whole tables at a time. If that takes too long, you can -try using the ``InlineVerifier``. Alternatively, you can verify in a staging -run and not verify during the production run (see :ref:`copydbinprod`). - -Note that the ``InlineVerifier`` on its own may potentially miss some -cases, and using it with the ``TargetVerifier`` is recommended if these -cases are possible. - -+---------------------------------------------------+---------------+---------------+-----------------+ -| Conditions | ChecksumTable | Inline | Inline + Target | -+---------------------------------------------------+---------------+---------------+-----------------+ -| Data inconsistency due to Ghostferry issuing an | Yes [3]_ | Yes | Yes | -| incorrect UPDATE on the target database (example: | | | | -| encoding-type issues). | | | | -+---------------------------------------------------+---------------+---------------+-----------------+ -| Data inconsistency due to Ghostferry failing to | Yes | Yes | Yes | -| INSERT on the target database. | | | | -+---------------------------------------------------+---------------+---------------+-----------------+ -| Data inconsistency due to Ghostferry failing to | Yes | Yes | Yes | -| DELETE on the target database. | | | | -+---------------------------------------------------+---------------+---------------+-----------------+ -| Data inconsistency due to rogue application | Yes | Sometimes [4]_| Yes | -| issuing writes (INSERT/UPDATE/DELETE) against the | | | | -| target database. | | | | -+---------------------------------------------------+---------------+---------------+-----------------+ -| Data inconsistency due to missing binlog events | Yes | Sometimes [5]_| Sometimes [5]_ | -| when Ghostferry is resumed from the wrong | | | | -| binlog coordinates. | | | | -+---------------------------------------------------+---------------+---------------+-----------------+ -| Data inconsistency if Ghostferry's Binlog writing | Yes | Probably not | Probably not | -| implementation is incorrect and modified the | | [6]_ | [6]_ | -| wrong row on the target (example, an UPDATE is | | | | -| supposed to go to id = 1 but Ghostferry instead | | | | -| issued a query for id = 2). This is an unrealistic| | | | -| scenario, but is included for illustrative | | | | -| purposes. | | | | -+---------------------------------------------------+---------------+---------------+-----------------+ - - -.. [3] Note that the CHECKSUM TABLE statement is broken in MySQL 5.7 for tables - with JSON columns. These tables will result in a false positive event: - even if two tables are identical, they can emit different checksums. See - https://bugs.mysql.com/bug.php?id=87847. This applies to every row in - this table. - -.. [4] If the rows modified by the rogue application are modified again on - the source after Ghostferry starts, the InlineVerifier's binlog tailer - should pick up that row and attempt to reverify it. - -.. [5] If the rows missed after resume are modified again on the source after - Ghostferry starts, the InlineVerifier's binlog tailer should pick up - that row and attempt to reverify it. - -.. [6] If the implementation of the Ghostferry algorithm is so broken, chances - are the InlineVerifier won't catch it either as it relies on the same - algorithm to enumerate the table and tail the binlogs. - -IterativeVerifier (Deprecated) ------------------------------- - -**NOTE! This is a deprecated verifier. Use the InlineVerifier instead.** - -IterativeVerifier verifies the source and target in a couple of steps: - -1. After the data copy, it first compares the hashes of each applicable rows - of the source and the target together to make sure they are the same. This - is known as the initial verification. - - a. If they are the same: the verification for that row is complete. - b. If they are not the same: add it into a reverify queue. - -2. For any rows changed during the initial verification process, add it into - the reverify queue. - -3. After the initial verification, verify the rows' hashes in the - reverification queue again. This is done to reduce the time needed to - reverify during the cutover as we assume the reverification queue will - become smaller during this process. - -4. During the cutover stage, verify all rows' hashes in the reverify queue. - - a. If they are the same: the verification for that row is complete. - b. If they are not the same: the verification fails. - -5. If no verification failure occurs, the source and the target are identical. - If verification failure does occur (4b), then the source and target are not - identical. - -A proof of concept TLA+ verification of this algorithm is done in -``_. - -InlineVerifier ------------------ - -InlineVerifier verifies the source and target inline with the other components with -a few slight differences from the IterativeVerifier above. The primary difference -being that this verification process happens while the data is being copied by the -DataIterator instead of after the fact. - -With regards to the ``DataIterator`` and ``BatchWriter``: - -1. While selecting the data in the ``DataIterator``, a fingerprint is appended - to the end of the statement that ``SELECT`` s data from the source as - ``SELECT *, MD5(...) FROM ...`` - -2. The fingerprint, gathered from the ``MD5(...)`` of the query above is stored - on the ``RowBatch`` to be used in the next verification step. - -3. The ``BatchWriter`` then attempts to write the ``RowBatch``, but instead of inserting - it directly, the following process is taken: - - a. A transaction is opened. - b. The data contained in the ``RowBatch`` is inserted. - c. The PK and fingerprint is then ``SELECT`` ed from the Target - as ``SELECT pk, MD5(....) FROM ...``. - d. The fingerprint (``MD5``) is then checked against the fingerprint currently - stored on the ``RowBatch``. - - The process in step 3 above is retried (with a limit) if there happens to be - a failure or mismatch, and will fail the run if they are not verified within - the retry limits. - -With regards to the BinlogStreamer: - -1. As DMLs are observed by the ``BinlogStreamer``, the PKs of the events are placed into - a ``reverifyStore`` to be periodically verified for correctness. - -2. This continues to happen in the background throughout the process of the Run. - -3. If a PK is found not to match, it is added back into the reverifyStore to be verified - again. - -4. When ``VerifyBeforeCutover`` starts, the InlineVerifier will verify enough of the - events in the ``reverifyStore`` to ensure it has a sufficiently small number of events - that can be successfully verified before cutover. - -5. When ``VerifyDuringCutover`` begins, all of the remaining events in the ``reverifyStore`` - are verified and any mismatches are returned. - -TargetVerifier ------------------ - -TargetVerifier ensures data on the Target is not corrupted during the move process -and is meant to be used in conjunction with another verifier above. - -It uses a configurable annotation string that is prepended to DMLs that acts as -a verified "signature" of all of Ghostferry's operations on the Target: - -1. A BinlogStreamer is created and attached to the Target - -2. As this BinlogStreamer receives DML events, it attempts to extract the annotation - from each for each of the ``RowsEvents``. - -3. If an annotation is not found for the DML, or the extracted annotation does not -match the configured annotation of Ghostferry, an error is returned and the process fails. - -The TargetVerifier needs to be manually stopped before cutover. If it is not stopped, -it may detect writes from the application (that are not from Ghostferry) and fail the run. -Stopping before cutover also gives the TargetVerifier the opportunity to inspect all -of the DMLs in its ``BinlogStreamer`` queue to ensure no corruption of the data has occurred. diff --git a/docs/source/technicaloverview.rst b/docs/technicaloverview.md similarity index 77% rename from docs/source/technicaloverview.rst rename to docs/technicaloverview.md index 27b576fdd..4d925c77e 100644 --- a/docs/source/technicaloverview.rst +++ b/docs/technicaloverview.md @@ -1,8 +1,6 @@ -.. _technicaloverview: + -================== -Technical Overview -================== +# Technical Overview Ghostferry is a Go library to move data from one MySQL instance to another while the source (and possibly the target) databases are online. In order to do @@ -53,8 +51,7 @@ This process has some downtime between step 5 and step 7. The window of downtime is proportional to how fast these steps can be done. In most cases this should be on the order of seconds to minutes. -Architecture ------------- +## Architecture Ghostferry has three levels of public APIs that you can use: the Ferry level, the DataIterator/BinlogStreamer level, and the Cursor level. Most of the time @@ -68,43 +65,40 @@ runs. The overall, simplified architecture of Ghostferry can be summarized with the figure below. It shows the basic flow of all the background tasks, along with -how each task is spawned (starting from ``Ferry.Run``). Arrows pointing towards +how each task is spawned (starting from `Ferry.Run`). Arrows pointing towards outside of an encapsulating box indicate the task will exit. The red arrows with "Error action" indicates an error has occurred and the error is sent to -the ``ErrorHandler``, at which point the ErrorHandler flow takes over. +the `ErrorHandler`, at which point the ErrorHandler flow takes over. -.. image:: _static/ghostferry-architecture.png - :align: center +![Ghostferry architecture](_static/ghostferry-architecture.png) You can see an example of an application built with Ghostferry in the -``copydb`` package. +`copydb` package. -Limitations ------------ +## Limitations - Right now, Ghostferry can only be used on tables with auto incrementing, numeric, and unique primary keys. - - An error will be emitted during the beginning of the run if such a primary - key is not detected. - - In the near future, we will extend support to arbitrary primary key types. - - To work around these restrictions, you can use mysqldump to dump and restore - the table during the cutover. + - An error will be emitted during the beginning of the run if such a primary + key is not detected. + - In the near future, we will extend support to arbitrary primary key types. + - To work around these restrictions, you can use mysqldump to dump and restore + the table during the cutover. - Ghostferry can only be used on a source database with FULL RBR. - - An error will be emitted during the beginning of the run if FULL RBR is - An error will be emitted during the beginning of the run if FULL RBR is not enabled on the source database. - - Without FULL RBR, the integrity of the data cannot be guaranteed. + - An error will be emitted during the beginning of the run if FULL RBR is + An error will be emitted during the beginning of the run if FULL RBR is not enabled on the source database. + - Without FULL RBR, the integrity of the data cannot be guaranteed. - Ghostferry does not support tables with foreign key constraints. - - For tables with foreign key constraints, the constraints should be removed - before performing the data migration. + - For tables with foreign key constraints, the constraints should be removed + before performing the data migration. -Algorithm Correctness ---------------------- +## Algorithm Correctness The overall algorithm of Ghostferry is specified in a TLA+ specification and -validated via TLC. The algorithm can be seen in the ``tlaplus`` directory in the +validated via TLC. The algorithm can be seen in the `tlaplus` directory in the source tree. diff --git a/docs/source/tutorialcopydb.rst b/docs/tutorialcopydb.md similarity index 53% rename from docs/source/tutorialcopydb.rst rename to docs/tutorialcopydb.md index 1a753b431..17d04dcf4 100644 --- a/docs/source/tutorialcopydb.rst +++ b/docs/tutorialcopydb.md @@ -1,8 +1,6 @@ -.. _tutorialcopydb: + -============================== -Tutorial for ghostferry-copydb -============================== +# Tutorial for ghostferry-copydb This tutorial aims to provide you with a first look on how to operate ghostferry-copydb to copy data from one database to another so you can have @@ -10,20 +8,19 @@ some experience with actually running Ghostferry. A production run of a data migration will be largely similar, although you will have to consider how to appropriately perform the cutover operations with respect to the applications accessing the database. Recommendations on how to run copydb in production can -be found in :ref:`copydbinprod`. +be found in [Running `ghostferry-copydb` in production](copydbinprod.md). -Setup and Seed MySQL --------------------- +## Setup and Seed MySQL In this tutorial, we will be using two test databases that we setup locally and we will not consider the application. With git, clone the Ghostferry repository and create the test MySQL instances: -.. code-block:: shell-session - - $ git clone https://github.com/Shopify/ghostferry.git - $ cd ghostferry - $ docker-compose up -d mysql-1 mysql-2 +```console +$ git clone https://github.com/Shopify/ghostferry.git +$ cd ghostferry +$ docker-compose up -d mysql-1 mysql-2 +``` Users without docker-compose can either install it on their machine or manually setup two localhost MySQL instances available at port 29291 and 29292 with FULL @@ -31,32 +28,31 @@ image row based replication. Confirm that you can access both MySQL instances with the MySQL console: -.. code-block:: shell-session - - # mysql --protocol=tcp -u root -P 29291 - # mysql --protocol=tcp -u root -P 29292 +```console +# mysql --protocol=tcp -u root -P 29291 +# mysql --protocol=tcp -u root -P 29292 +``` We will be moving data from the 29291 server to the 29292 server. To do this, we must first create some test data on the 29291 server to be copied over: -.. code-block:: shell-session - - # export LC_CTYPE=C # Only need this if you're mac - echo "CREATE DATABASE abc;" > /tmp/n1create.sql - echo "CREATE TABLE abc.table1 (id bigint(20) AUTO_INCREMENT, data varchar(16), primary key(id));" >> /tmp/n1create.sql - echo "CREATE TABLE abc.table2 (id bigint(20) AUTO_INCREMENT, data TEXT, primary key(id));" >> /tmp/n1create.sql - for i in `seq 1 350`; do - echo "INSERT INTO abc.table1 (id, data) VALUES (${i}, '$(cat /dev/urandom | tr -cd 'a-z0-9' | head -c 16)');" >> /tmp/n1create.sql - echo "INSERT INTO abc.table2 (id, data) VALUES (${i}, '$(cat /dev/urandom | tr -cd 'a-z0-9' | head -c 16)');" >> /tmp/n1create.sql - done - mysql --protocol=tcp -u root -P 29291 < /tmp/n1create.sql - rm /tmp/n1create.sql +```console +# export LC_CTYPE=C # Only need this if you're mac +echo "CREATE DATABASE abc;" > /tmp/n1create.sql +echo "CREATE TABLE abc.table1 (id bigint(20) AUTO_INCREMENT, data varchar(16), primary key(id));" >> /tmp/n1create.sql +echo "CREATE TABLE abc.table2 (id bigint(20) AUTO_INCREMENT, data TEXT, primary key(id));" >> /tmp/n1create.sql +for i in `seq 1 350`; do + echo "INSERT INTO abc.table1 (id, data) VALUES (${i}, '$(cat /dev/urandom | tr -cd 'a-z0-9' | head -c 16)');" >> /tmp/n1create.sql + echo "INSERT INTO abc.table2 (id, data) VALUES (${i}, '$(cat /dev/urandom | tr -cd 'a-z0-9' | head -c 16)');" >> /tmp/n1create.sql +done +mysql --protocol=tcp -u root -P 29291 < /tmp/n1create.sql +rm /tmp/n1create.sql +``` -This created two tables under the database ``abc``. We will be moving -``table1`` to 29292 while not copying 29291. +This created two tables under the database `abc`. We will be moving +`table1` to 29292 while not copying 29291. -(Mirrors Production) Create Ghostferry Users --------------------------------------------- +## (Mirrors Production) Create Ghostferry Users We then need to create a user with the appropriate permissions for Ghostferry to connect with, to perform the move on both servers. For this move, we @@ -65,28 +61,27 @@ production, you may want to enable that. On the source server, the minimum permissions required are: -.. code-block:: shell-session +```console +mysql> CREATE USER 'ghostferry'@'%' IDENTIFIED BY 'ghostferry'; +mysql> GRANT SELECT ON `abc`.* TO 'ghostferry'@'%'; +mysql> GRANT REPLICATION SLAVE, REPLICATION CLIENT ON *.* TO 'ghostferry'@'%'; +``` - mysql> CREATE USER 'ghostferry'@'%' IDENTIFIED BY 'ghostferry'; - mysql> GRANT SELECT ON `abc`.* TO 'ghostferry'@'%'; - mysql> GRANT REPLICATION SLAVE, REPLICATION CLIENT ON *.* TO 'ghostferry'@'%'; - -The above example grants the permission to only the ``abc`` database. You can +The above example grants the permission to only the `abc` database. You can grant it to more or all databases in your production environment as needed. On the target server, the minimum permissions required are: -.. code-block:: shell-session - - mysql> CREATE USER 'ghostferry'@'%' IDENTIFIED BY 'ghostferry'; - mysql> GRANT INSERT, UPDATE, DELETE, CREATE, SELECT ON *.* TO 'ghostferry'@'%'; +```console +mysql> CREATE USER 'ghostferry'@'%' IDENTIFIED BY 'ghostferry'; +mysql> GRANT INSERT, UPDATE, DELETE, CREATE, SELECT ON *.* TO 'ghostferry'@'%'; +``` -We grant permission to all databases because we assume that the ``abc`` +We grant permission to all databases because we assume that the `abc` database does not exist on the target and Ghostferry will create it automatically. -(Mirrors Production) Install ghostferry-copydb ----------------------------------------------- +## (Mirrors Production) Install ghostferry-copydb We then need to obtain the ghostferry-copydb binary on the server on which we want to execute Ghostferry on. Note that all the data moved will go through @@ -95,72 +90,69 @@ appropriately picked. For the present tutorial, Ghostferry will simply live on the same machine. To download the latest binaries, you currently have to compile copydb with -Go 1.9 via ``make copydb`` after cloning the repository. +Go 1.9 via `make copydb` after cloning the repository. -For testing purposes, you can also use `this unofficial PPA -`_ (see -`this PR `_ as well) to obtain a +For testing purposes, you can also use [this unofficial PPA](https://launchpad.net/~shuhao/+archive/ubuntu/ghostferry-unofficial) (see +[this PR](https://github.com/Shopify/ghostferry/pull/15) as well) to obtain a version of ghostferry-copydb. Note the unofficial PPA for ghsotferry-copydb is not supported and you should not use it in production. -(Mirrors Production) Setup Ghostferry Run Configuration -------------------------------------------------------- +## (Mirrors Production) Setup Ghostferry Run Configuration We will need to provide ghostferry-copydb with a configuration file such that it knows how to connect to the databases and what to copy. This is a json file which should look like the following: -.. code-block:: json - - { - "Source": { - "Host": "127.0.0.1", - "Port": 29291, - "User": "ghostferry", - "Pass": "ghostferry", - "Collation": "utf8mb4_unicode_ci", - "Params": { - "charset": "utf8mb4" - } - }, - - "Target": { - "Host": "127.0.0.1", - "Port": 29292, - "User": "ghostferry", - "Pass": "ghostferry", - "Collation": "utf8mb4_unicode_ci", - "Params": { - "charset": "utf8mb4" - } - }, - - "Databases": { - "Whitelist": ["abc"] - }, - - "Tables": { - "Blacklist": ["table2"] - }, - - "VerifierType": "ChecksumTable" - } - -Save this file to a file called ``examplerun.json``. +```json +{ + "Source": { + "Host": "127.0.0.1", + "Port": 29291, + "User": "ghostferry", + "Pass": "ghostferry", + "Collation": "utf8mb4_unicode_ci", + "Params": { + "charset": "utf8mb4" + } + }, + + "Target": { + "Host": "127.0.0.1", + "Port": 29292, + "User": "ghostferry", + "Pass": "ghostferry", + "Collation": "utf8mb4_unicode_ci", + "Params": { + "charset": "utf8mb4" + } + }, + + "Databases": { + "Whitelist": ["abc"] + }, + + "Tables": { + "Blacklist": ["table2"] + }, + + "VerifierType": "ChecksumTable" +} +``` + +Save this file to a file called `examplerun.json`. Note that in the example above, the Collation and charsets are set. If you setup your own MySQL instances, you might need to change these values. We are -also using the ``Whitelist`` and ``Blacklist`` to ensure that we only copy -``abc.table1`` from the source to the target. For more information about this -configuration file, see :ref:`copydbinprod`. +also using the `Whitelist` and `Blacklist` to ensure that we only copy +`abc.table1` from the source to the target. For more information about this +configuration file, see [Running `ghostferry-copydb` in production](copydbinprod.md). Lastly, we have enabled verification to be available to use during the run. Specifically, we enabled the ChecksumTable verifier as the amount of data copied will be small. For more information about the verifiers, see -:ref:`verifiers`. +[Verifiers](verifiers.md). -(Mirrors Production) Validate Ghostferry Configuration ------------------------------------------------------- +## (Mirrors Production) Validate Ghostferry Configuration Before actually running Ghostferry, it is good practise to validate the configuration you specified. ghostferry-copydb has a dryrun flag that will try @@ -168,60 +160,58 @@ to use the configuration you have to connect to the database. It will also scan the tables according to the black/whitelist specified and print it out in the debug logs: -.. code-block:: shell-session - - $ ghostferry-copydb -dryrun -verbose examplerun.json +```console +$ ghostferry-copydb -dryrun -verbose examplerun.json +``` The verbose flag gives slightly more debug information in case there are any issues. In this case, there should not be any issues as we setup the database according to the tutorial and the output should be something like this (simplified for readibility in the tutorial): -.. code-block:: text - - [...] - INFO[0000] connecting to the source database dsn="ghostferry:@[...]" tag=ferry - INFO[0000] connecting to the target database dsn="ghostferry:@[...]" tag=ferry - [...] - INFO[0000] found binlog position, starting synchronization file=[...] pos=[...] tag=binlog_streamer - [...] - DEBU[0000] loading tables from database database=abc tag=table_schema_cache - DEBU[0000] fetching table schema database=abc table=table1 tag=table_schema_cache - DEBU[0000] fetching table schema database=abc table=table2 tag=table_schema_cache - DEBU[0000] caching table schema database=abc table=table1 tag=table_schema_cache - INFO[0000] table schemas cached tables="[abc.table1]" tag=table_schema_cache - exiting due to dryrun +```text +[...] +INFO[0000] connecting to the source database dsn="ghostferry:@[...]" tag=ferry +INFO[0000] connecting to the target database dsn="ghostferry:@[...]" tag=ferry +[...] +INFO[0000] found binlog position, starting synchronization file=[...] pos=[...] tag=binlog_streamer +[...] +DEBU[0000] loading tables from database database=abc tag=table_schema_cache +DEBU[0000] fetching table schema database=abc table=table1 tag=table_schema_cache +DEBU[0000] fetching table schema database=abc table=table2 tag=table_schema_cache +DEBU[0000] caching table schema database=abc table=table1 tag=table_schema_cache +INFO[0000] table schemas cached tables="[abc.table1]" tag=table_schema_cache +exiting due to dryrun +``` Note the last INFO line shows which tables will be moved as we cache their schemas in the memory. If there is a table you want to move and it does not show up there, it means the whitelist/blacklist configuration is incorrect. -(Mirrors Production) Starting Ghostferry Run --------------------------------------------- +## (Mirrors Production) Starting Ghostferry Run To start the ghostferry run, simply perform the same command as before except without the dryrun flag. You can also turn off the verbose flag, although it may be good practise to leave it on and redirect stdout to a file so the move can be audited at a later time. We will do this here for good practise: -.. code-block:: shell-session - - $ ghostferry-copydb -verbose examplerun.json 2&>examplerun.log +```console +$ ghostferry-copydb -verbose examplerun.json 2&>examplerun.log +``` To confirm that Ghostferry indeed copies changes to the source table, we can -manually insert a row into ``abc.table1`` during the run - -.. code-block:: shell-session +manually insert a row into `abc.table1` during the run - # mysql --protocol=tcp -u root -P 29291 - mysql> INSERT INTO abc.table1 (id, data) VALUES (351, "helloworld"); +```console +# mysql --protocol=tcp -u root -P 29291 +mysql> INSERT INTO abc.table1 (id, data) VALUES (351, "helloworld"); +``` -(Mirrors Production) Monitoring Ghostferry Run via Web UI ---------------------------------------------------------- +## (Mirrors Production) Monitoring Ghostferry Run via Web UI Once the run starts, a built-in webserver is started at port 8000 by default. This can be changed in the configuration json. Simply browse to -http://localhost:8000 to view this server and in there you should find controls + to view this server and in there you should find controls to: - Pause/Unpause: allows you to pause/unpause the data copy and binlog streaming @@ -246,8 +236,7 @@ For this tutorial, the run should be very short so thus you might miss most of the copying states. Take a look around and refresh a couple times to get familiar with the UI. -(Mirrors Production) Perform Cutover ------------------------------------- +## (Mirrors Production) Perform Cutover In the default configuration, cutover is triggered manually. During cutover, you must stop writes to the data on the source database. For the purpose of @@ -255,26 +244,25 @@ this tutorial, we will set the source database to read only. Even though we have no applications writing to the source in this case, let's do it anyway so we get into the habit of thinking of this step: -.. code-block:: shell-session +```console +# mysql --protocol=tcp -u root -P 29291 +mysql> FLUSH TABLES WITH READ LOCK; -- Ensure all writes are done +mysql> SET GLOBAL read_only = ON; -- Sets the database to read only +mysql> FLUSH BINARY LOGS -- Ensure all writes are record in binlog +``` - # mysql --protocol=tcp -u root -P 29291 - mysql> FLUSH TABLES WITH READ LOCK; -- Ensure all writes are done - mysql> SET GLOBAL read_only = ON; -- Sets the database to read only - mysql> FLUSH BINARY LOGS -- Ensure all writes are record in binlog - -The last step ``FLUSH BINARY LOGS`` is not necessarily required if you run your -MySQL server with ``sync_binlog=1``. If you're running Ghostferry from a source -that is a replica, you need to also turn on the option ``RunFerryFromReplica`` +The last step `FLUSH BINARY LOGS` is not necessarily required if you run your +MySQL server with `sync_binlog=1`. If you're running Ghostferry from a source +that is a replica, you need to also turn on the option `RunFerryFromReplica` in the config json as well as other options. See -``_ for more + for more details. We can then go back to the web ui and click the Allow Automatic Cutover button. In a second or two the ghostferry binlog streaming process should stop. Refresh the page until you see the state to be DONE. -(Mirrors Production) Verify Source and Target Data are Identical ----------------------------------------------------------------- +## (Mirrors Production) Verify Source and Target Data are Identical At this point, the data on the source and target should be identical. To confirm this is the case, click the Run Verification button in the web ui to @@ -284,13 +272,12 @@ until it tells you the verification was successful. Additionally, since we manually inserted a row earlier, we should be able to find it via: -.. code-block:: shell-session - - # mysql --protocol=tcp -u root -P 29292 - mysql> SELECT * FROM abc.table1 WHERE id = 351; +```console +# mysql --protocol=tcp -u root -P 29292 +mysql> SELECT * FROM abc.table1 WHERE id = 351; +``` -Finishing Ghostferry Run and Next Steps ---------------------------------------- +## Finishing Ghostferry Run and Next Steps At this point, the data on the source and target are verified identical and Ghostferry will no longer propagate data from 29291 to 29292. In a production @@ -300,6 +287,7 @@ the target database. The control server UI will stay up indefinitely. To stop it, simply press CTRL+C to interrupt the ghostferry-copydb process. -To run Ghostferry in production, you should read through :ref:`copydbinprod`. +To run Ghostferry in production, you should read through +[Running `ghostferry-copydb` in production](copydbinprod.md). If you need to interrupt and resume Ghostferry, you should also read through -:ref:`copydbinterruptresume`. +[Interrupt and resuming `ghostferry-copydb`](copydbinterruptresume.md). diff --git a/docs/verifiers.md b/docs/verifiers.md new file mode 100644 index 000000000..b0eef7044 --- /dev/null +++ b/docs/verifiers.md @@ -0,0 +1,160 @@ +# Verifiers + +Verifiers in Ghostferry are designed to ensure that Ghostferry did not +corrupt/miss data. There are three different verifiers: the +`ChecksumTableVerifier`, the `InlineVerifier`, and the `TargetVerifier`. A comparison of the +`ChecksumTableVerifier` and `InlineVerifier` are given below: + +| | ChecksumTableVerifier | InlineVerifier | +|---|---|---| +| Mechanism | `CHECKSUM TABLE` | Verify row after insert; Reverify changed rows before and during cutover. | +| Impacts on Cutover Time | Linear w.r.t data size | Linear w.r.t. change rate [^1] | +| Impacts on Copy Time [^2] | None | Linear w.r.t data size | +| Memory Usage | Minimal | Linear w.r.t rows changed | +| Partial table copy | Not supported | Supported | +| Worst Case Scenario | Large databases causes unacceptable downtime | Verification is slower than the change rate of the DB | + +[^1]: Additional improvements could be made to reduce this as long as + Ghostferry is faster than the rate of change. See + . + +[^2]: Increase in copy time does not increase downtime. Downtime occurs only + in cutover. + +If you want verification, you should try with the `ChecksumTableVerifier` +first if you're copying whole tables at a time. If that takes too long, you can +try using the `InlineVerifier`. Alternatively, you can verify in a staging +run and not verify during the production run (see +[Running `ghostferry-copydb` in production](copydbinprod.md)). + +Note that the `InlineVerifier` on its own may potentially miss some +cases, and using it with the `TargetVerifier` is recommended if these +cases are possible. + +| Conditions | ChecksumTable | Inline | Inline + Target | +|---|---|---|---| +| Data inconsistency due to Ghostferry issuing an incorrect UPDATE on the target database (example: encoding-type issues). | Yes [^3] | Yes | Yes | +| Data inconsistency due to Ghostferry failing to INSERT on the target database. | Yes | Yes | Yes | +| Data inconsistency due to Ghostferry failing to DELETE on the target database. | Yes | Yes | Yes | +| Data inconsistency due to rogue application issuing writes (INSERT/UPDATE/DELETE) against the target database. | Yes | Sometimes [^4] | Yes | +| Data inconsistency due to missing binlog events when Ghostferry is resumed from the wrong binlog coordinates. | Yes | Sometimes [^5] | Sometimes [^5] | +| Data inconsistency if Ghostferry's Binlog writing implementation is incorrect and modified the wrong row on the target (example, an UPDATE is supposed to go to id = 1 but Ghostferry instead issued a query for id = 2). This is an unrealistic scenario, but is included for illustrative purposes. | Yes | Probably not [^6] | Probably not [^6] | + +[^3]: Note that the CHECKSUM TABLE statement is broken in MySQL 5.7 for tables + with JSON columns. These tables will result in a false positive event: + even if two tables are identical, they can emit different checksums. See + . This applies to every row in + this table. + +[^4]: If the rows modified by the rogue application are modified again on + the source after Ghostferry starts, the InlineVerifier's binlog tailer + should pick up that row and attempt to reverify it. + +[^5]: If the rows missed after resume are modified again on the source after + Ghostferry starts, the InlineVerifier's binlog tailer should pick up + that row and attempt to reverify it. + +[^6]: If the implementation of the Ghostferry algorithm is so broken, chances + are the InlineVerifier won't catch it either as it relies on the same + algorithm to enumerate the table and tail the binlogs. + +## IterativeVerifier (Deprecated) + +**NOTE! This is a deprecated verifier. Use the InlineVerifier instead.** + +IterativeVerifier verifies the source and target in a couple of steps: + +1. After the data copy, it first compares the hashes of each applicable rows + of the source and the target together to make sure they are the same. This + is known as the initial verification. + + 1. If they are the same: the verification for that row is complete. + 2. If they are not the same: add it into a reverify queue. + +2. For any rows changed during the initial verification process, add it into + the reverify queue. + +3. After the initial verification, verify the rows' hashes in the + reverification queue again. This is done to reduce the time needed to + reverify during the cutover as we assume the reverification queue will + become smaller during this process. + +4. During the cutover stage, verify all rows' hashes in the reverify queue. + + 1. If they are the same: the verification for that row is complete. + 2. If they are not the same: the verification fails. + +5. If no verification failure occurs, the source and the target are identical. + If verification failure does occur (4b), then the source and target are not + identical. + +A proof of concept TLA+ verification of this algorithm is done in +. + +## InlineVerifier + +InlineVerifier verifies the source and target inline with the other components with +a few slight differences from the IterativeVerifier above. The primary difference +being that this verification process happens while the data is being copied by the +DataIterator instead of after the fact. + +With regards to the `DataIterator` and `BatchWriter`: + +1. While selecting the data in the `DataIterator`, a fingerprint is appended + to the end of the statement that `SELECT`s data from the source as + `SELECT *, MD5(...) FROM ...` + +2. The fingerprint, gathered from the `MD5(...)` of the query above is stored + on the `RowBatch` to be used in the next verification step. + +3. The `BatchWriter` then attempts to write the `RowBatch`, but instead of inserting + it directly, the following process is taken: + + 1. A transaction is opened. + 2. The data contained in the `RowBatch` is inserted. + 3. The PK and fingerprint is then `SELECT`ed from the Target + as `SELECT pk, MD5(....) FROM ...`. + 4. The fingerprint (`MD5`) is then checked against the fingerprint currently + stored on the `RowBatch`. + + The process in step 3 above is retried (with a limit) if there happens to be + a failure or mismatch, and will fail the run if they are not verified within + the retry limits. + +With regards to the BinlogStreamer: + +1. As DMLs are observed by the `BinlogStreamer`, the PKs of the events are placed into + a `reverifyStore` to be periodically verified for correctness. + +2. This continues to happen in the background throughout the process of the Run. + +3. If a PK is found not to match, it is added back into the reverifyStore to be verified + again. + +4. When `VerifyBeforeCutover` starts, the InlineVerifier will verify enough of the + events in the `reverifyStore` to ensure it has a sufficiently small number of events + that can be successfully verified before cutover. + +5. When `VerifyDuringCutover` begins, all of the remaining events in the `reverifyStore` + are verified and any mismatches are returned. + +## TargetVerifier + +TargetVerifier ensures data on the Target is not corrupted during the move process +and is meant to be used in conjunction with another verifier above. + +It uses a configurable annotation string that is prepended to DMLs that acts as +a verified "signature" of all of Ghostferry's operations on the Target: + +1. A BinlogStreamer is created and attached to the Target + +2. As this BinlogStreamer receives DML events, it attempts to extract the annotation + from each for each of the `RowsEvents`. + +3. If an annotation is not found for the DML, or the extracted annotation does not + match the configured annotation of Ghostferry, an error is returned and the process fails. + +The TargetVerifier needs to be manually stopped before cutover. If it is not stopped, +it may detect writes from the application (that are not from Ghostferry) and fail the run. +Stopping before cutover also gives the TargetVerifier the opportunity to inspect all +of the DMLs in its `BinlogStreamer` queue to ensure no corruption of the data has occurred.