chain_indexer.go 17.2 KB
Newer Older
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19
// Copyright 2017 The go-ethereum Authors
// This file is part of the go-ethereum library.
//
// The go-ethereum library is free software: you can redistribute it and/or modify
// it under the terms of the GNU Lesser General Public License as published by
// the Free Software Foundation, either version 3 of the License, or
// (at your option) any later version.
//
// The go-ethereum library is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Lesser General Public License for more details.
//
// You should have received a copy of the GNU Lesser General Public License
// along with the go-ethereum library. If not, see <http://www.gnu.org/licenses/>.

package core

import (
20
	"context"
21
	"encoding/binary"
22
	"fmt"
23
	"sync"
24
	"sync/atomic"
25 26 27
	"time"

	"github.com/ethereum/go-ethereum/common"
28
	"github.com/ethereum/go-ethereum/core/rawdb"
29 30 31
	"github.com/ethereum/go-ethereum/core/types"
	"github.com/ethereum/go-ethereum/ethdb"
	"github.com/ethereum/go-ethereum/event"
32
	"github.com/ethereum/go-ethereum/log"
33 34
)

35 36 37 38 39 40
// ChainIndexerBackend defines the methods needed to process chain segments in
// the background and write the segment results into the database. These can be
// used to create filter blooms or CHTs.
type ChainIndexerBackend interface {
	// Reset initiates the processing of a new chain segment, potentially terminating
	// any partially completed operations (in case of a reorg).
41
	Reset(ctx context.Context, section uint64, prevHead common.Hash) error
42 43 44

	// Process crunches through the next header in the chain segment. The caller
	// will ensure a sequential order of headers.
45
	Process(ctx context.Context, header *types.Header) error
46

47
	// Commit finalizes the section metadata and stores it into the database.
48
	Commit() error
49 50 51

	// Prune deletes the chain index older than the given threshold.
	Prune(threshold uint64) error
52 53
}

54 55 56 57 58
// ChainIndexerChain interface is used for connecting the indexer to a blockchain
type ChainIndexerChain interface {
	// CurrentHeader retrieves the latest locally known header.
	CurrentHeader() *types.Header

59 60
	// SubscribeChainHeadEvent subscribes to new head header notifications.
	SubscribeChainHeadEvent(ch chan<- ChainHeadEvent) event.Subscription
61 62
}

63 64 65
// ChainIndexer does a post-processing job for equally sized sections of the
// canonical chain (like BlooomBits and CHT structures). A ChainIndexer is
// connected to the blockchain through the event system by starting a
66
// ChainHeadEventLoop in a goroutine.
67 68 69 70 71 72 73 74 75 76 77
//
// Further child ChainIndexers can be added which use the output of the parent
// section indexer. These child indexers receive new head notifications only
// after an entire section has been finished or in case of rollbacks that might
// affect already finished sections.
type ChainIndexer struct {
	chainDb  ethdb.Database      // Chain database to index the data from
	indexDb  ethdb.Database      // Prefixed table-view of the db to write index metadata into
	backend  ChainIndexerBackend // Background processor generating the index data content
	children []*ChainIndexer     // Child indexers to cascade chain updates to

78 79 80 81 82
	active    uint32          // Flag whether the event loop was started
	update    chan struct{}   // Notification channel that headers should be processed
	quit      chan chan error // Quit channel to tear down running goroutines
	ctx       context.Context
	ctxCancel func()
83 84 85 86 87 88 89 90

	sectionSize uint64 // Number of blocks in a single chain segment to process
	confirmsReq uint64 // Number of confirmations before processing a completed segment

	storedSections uint64 // Number of sections successfully indexed into the database
	knownSections  uint64 // Number of sections known to be complete (block wise)
	cascadedHead   uint64 // Block number of the last completed section cascaded to subindexers

91 92 93
	checkpointSections uint64      // Number of sections covered by the checkpoint
	checkpointHead     common.Hash // Section head belonging to the checkpoint

94 95 96
	throttling time.Duration // Disk throttling to prevent a heavy upgrade from hogging resources

	log  log.Logger
97
	lock sync.Mutex
98 99
}

100 101 102
// NewChainIndexer creates a new chain indexer to do background processing on
// chain segments of a given size after certain number of confirmations passed.
// The throttling parameter might be used to prevent database thrashing.
103
func NewChainIndexer(chainDb ethdb.Database, indexDb ethdb.Database, backend ChainIndexerBackend, section, confirm uint64, throttling time.Duration, kind string) *ChainIndexer {
104 105 106 107
	c := &ChainIndexer{
		chainDb:     chainDb,
		indexDb:     indexDb,
		backend:     backend,
108 109 110 111 112 113
		update:      make(chan struct{}, 1),
		quit:        make(chan chan error),
		sectionSize: section,
		confirmsReq: confirm,
		throttling:  throttling,
		log:         log.New("type", kind),
114
	}
115 116
	// Initialize database dependent fields and start the updater
	c.loadValidSections()
117 118
	c.ctx, c.ctxCancel = context.WithCancel(context.Background())

119
	go c.updateLoop()
120

121 122 123
	return c
}

124 125 126 127 128 129 130
// AddCheckpoint adds a checkpoint. Sections are never processed and the chain
// is not expected to be available before this point. The indexer assumes that
// the backend has sufficient information available to process subsequent sections.
//
// Note: knownSections == 0 and storedSections == checkpointSections until
// syncing reaches the checkpoint
func (c *ChainIndexer) AddCheckpoint(section uint64, shead common.Hash) {
131 132 133
	c.lock.Lock()
	defer c.lock.Unlock()

134 135 136 137
	// Short circuit if the given checkpoint is below than local's.
	if c.checkpointSections >= section+1 || section < c.storedSections {
		return
	}
138 139 140
	c.checkpointSections = section + 1
	c.checkpointHead = shead

141 142 143 144
	c.setSectionHead(section, shead)
	c.setValidSections(section + 1)
}

145
// Start creates a goroutine to feed chain head events into the indexer for
146 147
// cascading background processing. Children do not need to be started, they
// are notified about new events by their parents.
148
func (c *ChainIndexer) Start(chain ChainIndexerChain) {
149 150
	events := make(chan ChainHeadEvent, 10)
	sub := chain.SubscribeChainHeadEvent(events)
151

152
	go c.eventLoop(chain.CurrentHeader(), events, sub)
153
}
154

155 156 157 158
// Close tears down all goroutines belonging to the indexer and returns any error
// that might have occurred internally.
func (c *ChainIndexer) Close() error {
	var errs []error
159

160 161
	c.ctxCancel()

162 163 164 165 166 167 168 169 170 171 172 173 174
	// Tear down the primary update loop
	errc := make(chan error)
	c.quit <- errc
	if err := <-errc; err != nil {
		errs = append(errs, err)
	}
	// If needed, tear down the secondary event loop
	if atomic.LoadUint32(&c.active) != 0 {
		c.quit <- errc
		if err := <-errc; err != nil {
			errs = append(errs, err)
		}
	}
175 176 177 178 179 180
	// Close all children
	for _, child := range c.children {
		if err := child.Close(); err != nil {
			errs = append(errs, err)
		}
	}
181 182 183 184
	// Return any failures
	switch {
	case len(errs) == 0:
		return nil
185

186 187
	case len(errs) == 1:
		return errs[0]
188

189 190
	default:
		return fmt.Errorf("%v", errs)
191 192 193
	}
}

194 195 196
// eventLoop is a secondary - optional - event loop of the indexer which is only
// started for the outermost indexer to push chain head events into a processing
// queue.
197
func (c *ChainIndexer) eventLoop(currentHeader *types.Header, events chan ChainHeadEvent, sub event.Subscription) {
198 199 200 201 202 203
	// Mark the chain indexer as active, requiring an additional teardown
	atomic.StoreUint32(&c.active, 1)

	defer sub.Unsubscribe()

	// Fire the initial new head event to start any outstanding processing
204
	c.newHead(currentHeader.Number.Uint64(), false)
205 206 207 208 209

	var (
		prevHeader = currentHeader
		prevHash   = currentHeader.Hash()
	)
210 211
	for {
		select {
212 213 214
		case errc := <-c.quit:
			// Chain indexer terminating, report no failure and abort
			errc <- nil
215
			return
216

217
		case ev, ok := <-events:
218 219 220 221 222 223
			// Received a new event, ensure it's not nil (closing) and update
			if !ok {
				errc := <-c.quit
				errc <- nil
				return
			}
224
			header := ev.Block.Header()
225
			if header.ParentHash != prevHash {
226
				// Reorg to the common ancestor if needed (might not exist in light sync mode, skip reorg then)
227
				// TODO(karalabe, zsfelfoldi): This seems a bit brittle, can we detect this case explicitly?
228

229 230 231 232
				if rawdb.ReadCanonicalHash(c.chainDb, prevHeader.Number.Uint64()) != prevHash {
					if h := rawdb.FindCommonAncestor(c.chainDb, prevHeader, header); h != nil {
						c.newHead(h.Number.Uint64(), true)
					}
233
				}
234 235 236 237
			}
			c.newHead(header.Number.Uint64(), false)

			prevHeader, prevHash = header, header.Hash()
238 239 240 241
		}
	}
}

242 243
// newHead notifies the indexer about new chain heads and/or reorgs.
func (c *ChainIndexer) newHead(head uint64, reorg bool) {
244 245 246
	c.lock.Lock()
	defer c.lock.Unlock()

247 248 249
	// If a reorg happened, invalidate all sections until that point
	if reorg {
		// Revert the known section number to the reorg point
250
		known := (head + 1) / c.sectionSize
251 252 253 254 255 256 257 258 259
		stored := known
		if known < c.checkpointSections {
			known = 0
		}
		if stored < c.checkpointSections {
			stored = c.checkpointSections
		}
		if known < c.knownSections {
			c.knownSections = known
260
		}
261
		// Revert the stored sections from the database to the reorg point
262 263
		if stored < c.storedSections {
			c.setValidSections(stored)
264
		}
265
		// Update the new head number to the finalized section end and notify children
266
		head = known * c.sectionSize
267

268 269 270 271
		if head < c.cascadedHead {
			c.cascadedHead = head
			for _, child := range c.children {
				child.newHead(c.cascadedHead, true)
272 273
			}
		}
274 275 276 277 278 279
		return
	}
	// No reorg, calculate the number of newly known sections and update if high enough
	var sections uint64
	if head >= c.confirmsReq {
		sections = (head + 1 - c.confirmsReq) / c.sectionSize
280 281 282
		if sections < c.checkpointSections {
			sections = 0
		}
283
		if sections > c.knownSections {
284 285 286 287 288 289 290 291
			if c.knownSections < c.checkpointSections {
				// syncing reached the checkpoint, verify section head
				syncedHead := rawdb.ReadCanonicalHash(c.chainDb, c.checkpointSections*c.sectionSize-1)
				if syncedHead != c.checkpointHead {
					c.log.Error("Synced chain does not match checkpoint", "number", c.checkpointSections*c.sectionSize-1, "expected", c.checkpointHead, "synced", syncedHead)
					return
				}
			}
292 293 294 295 296 297 298 299 300 301 302 303 304
			c.knownSections = sections

			select {
			case c.update <- struct{}{}:
			default:
			}
		}
	}
}

// updateLoop is the main event loop of the indexer which pushes chain segments
// down into the processing backend.
func (c *ChainIndexer) updateLoop() {
305
	var (
306 307
		updating bool
		updated  time.Time
308
	)
309

310 311 312 313 314 315 316 317 318 319 320 321 322 323
	for {
		select {
		case errc := <-c.quit:
			// Chain indexer terminating, report no failure and abort
			errc <- nil
			return

		case <-c.update:
			// Section headers completed (or rolled back), update the index
			c.lock.Lock()
			if c.knownSections > c.storedSections {
				// Periodically print an upgrade log message to the user
				if time.Since(updated) > 8*time.Second {
					if c.knownSections > c.storedSections+1 {
324
						updating = true
325 326 327 328 329
						c.log.Info("Upgrading chain index", "percentage", c.storedSections*100/c.knownSections)
					}
					updated = time.Now()
				}
				// Cache the current section count and head to allow unlocking the mutex
330
				c.verifyLastHead()
331 332 333
				section := c.storedSections
				var oldHead common.Hash
				if section > 0 {
334
					oldHead = c.SectionHead(section - 1)
335 336 337 338
				}
				// Process the newly defined section in the background
				c.lock.Unlock()
				newHead, err := c.processSection(section, oldHead)
339
				if err != nil {
340 341 342 343 344 345
					select {
					case <-c.ctx.Done():
						<-c.quit <- nil
						return
					default:
					}
346 347
					c.log.Error("Section processing failed", "error", err)
				}
348 349
				c.lock.Lock()

350 351
				// If processing succeeded and no reorgs occurred, mark the section completed
				if err == nil && (section == 0 || oldHead == c.SectionHead(section-1)) {
352 353
					c.setSectionHead(section, newHead)
					c.setValidSections(section + 1)
354 355
					if c.storedSections == c.knownSections && updating {
						updating = false
356 357
						c.log.Info("Finished upgrading chain index")
					}
358 359 360 361 362 363 364 365
					c.cascadedHead = c.storedSections*c.sectionSize - 1
					for _, child := range c.children {
						c.log.Trace("Cascading chain index update", "head", c.cascadedHead)
						child.newHead(c.cascadedHead, false)
					}
				} else {
					// If processing failed, don't retry until further notification
					c.log.Debug("Chain index processing failed", "section", section, "err", err)
366
					c.verifyLastHead()
367
					c.knownSections = c.storedSections
368 369
				}
			}
370 371 372 373 374 375 376 377 378 379
			// If there are still further sections to process, reschedule
			if c.knownSections > c.storedSections {
				time.AfterFunc(c.throttling, func() {
					select {
					case c.update <- struct{}{}:
					default:
					}
				})
			}
			c.lock.Unlock()
380 381 382 383
		}
	}
}

384 385 386 387 388 389 390 391
// processSection processes an entire section by calling backend functions while
// ensuring the continuity of the passed headers. Since the chain mutex is not
// held while processing, the continuity can be broken by a long reorg, in which
// case the function returns with an error.
func (c *ChainIndexer) processSection(section uint64, lastHead common.Hash) (common.Hash, error) {
	c.log.Trace("Processing new chain section", "section", section)

	// Reset and partial processing
392
	if err := c.backend.Reset(c.ctx, section, lastHead); err != nil {
393 394 395
		c.setValidSections(0)
		return common.Hash{}, err
	}
396

397
	for number := section * c.sectionSize; number < (section+1)*c.sectionSize; number++ {
398
		hash := rawdb.ReadCanonicalHash(c.chainDb, number)
399
		if hash == (common.Hash{}) {
400
			return common.Hash{}, fmt.Errorf("canonical block #%d unknown", number)
401
		}
402
		header := rawdb.ReadHeader(c.chainDb, hash, number)
403 404 405 406
		if header == nil {
			return common.Hash{}, fmt.Errorf("block #%d [%x…] not found", number, hash[:4])
		} else if header.ParentHash != lastHead {
			return common.Hash{}, fmt.Errorf("chain reorged during section processing")
407
		}
408 409 410
		if err := c.backend.Process(c.ctx, header); err != nil {
			return common.Hash{}, err
		}
411
		lastHead = header.Hash()
412
	}
413
	if err := c.backend.Commit(); err != nil {
414
		return common.Hash{}, err
415
	}
416
	return lastHead, nil
417 418
}

419 420 421 422
// verifyLastHead compares last stored section head with the corresponding block hash in the
// actual canonical chain and rolls back reorged sections if necessary to ensure that stored
// sections are all valid
func (c *ChainIndexer) verifyLastHead() {
423
	for c.storedSections > 0 && c.storedSections > c.checkpointSections {
424 425 426 427 428 429 430
		if c.SectionHead(c.storedSections-1) == rawdb.ReadCanonicalHash(c.chainDb, c.storedSections*c.sectionSize-1) {
			return
		}
		c.setValidSections(c.storedSections - 1)
	}
}

431 432 433 434
// Sections returns the number of processed sections maintained by the indexer
// and also the information about the last header indexed for potential canonical
// verifications.
func (c *ChainIndexer) Sections() (uint64, uint64, common.Hash) {
435 436 437
	c.lock.Lock()
	defer c.lock.Unlock()

438
	c.verifyLastHead()
439
	return c.storedSections, c.storedSections*c.sectionSize - 1, c.SectionHead(c.storedSections - 1)
440 441 442 443
}

// AddChildIndexer adds a child ChainIndexer that can use the output of this one
func (c *ChainIndexer) AddChildIndexer(indexer *ChainIndexer) {
444 445 446
	if indexer == c {
		panic("can't add indexer as a child of itself")
	}
447 448 449 450 451 452
	c.lock.Lock()
	defer c.lock.Unlock()

	c.children = append(c.children, indexer)

	// Cascade any pending updates to new children too
453 454 455 456 457 458 459 460
	sections := c.storedSections
	if c.knownSections < sections {
		// if a section is "stored" but not "known" then it is a checkpoint without
		// available chain data so we should not cascade it yet
		sections = c.knownSections
	}
	if sections > 0 {
		indexer.newHead(sections*c.sectionSize-1, false)
461 462 463
	}
}

464 465 466 467 468
// Prune deletes all chain data older than given threshold.
func (c *ChainIndexer) Prune(threshold uint64) error {
	return c.backend.Prune(threshold)
}

469 470 471
// loadValidSections reads the number of valid sections from the index database
// and caches is into the local state.
func (c *ChainIndexer) loadValidSections() {
472 473
	data, _ := c.indexDb.Get([]byte("count"))
	if len(data) == 8 {
474
		c.storedSections = binary.BigEndian.Uint64(data)
475 476 477 478
	}
}

// setValidSections writes the number of valid sections to the index database
479 480
func (c *ChainIndexer) setValidSections(sections uint64) {
	// Set the current number of valid sections in the database
481
	var data [8]byte
482
	binary.BigEndian.PutUint64(data[:], sections)
483
	c.indexDb.Put([]byte("count"), data[:])
484 485 486 487 488 489 490

	// Remove any reorged sections, caching the valids in the mean time
	for c.storedSections > sections {
		c.storedSections--
		c.removeSectionHead(c.storedSections)
	}
	c.storedSections = sections // needed if new > old
491 492
}

493
// SectionHead retrieves the last block hash of a processed section from the
494
// index database.
495
func (c *ChainIndexer) SectionHead(section uint64) common.Hash {
496
	var data [8]byte
497
	binary.BigEndian.PutUint64(data[:], section)
498 499 500 501 502 503 504 505

	hash, _ := c.indexDb.Get(append([]byte("shead"), data[:]...))
	if len(hash) == len(common.Hash{}) {
		return common.BytesToHash(hash)
	}
	return common.Hash{}
}

506 507 508
// setSectionHead writes the last block hash of a processed section to the index
// database.
func (c *ChainIndexer) setSectionHead(section uint64, hash common.Hash) {
509
	var data [8]byte
510
	binary.BigEndian.PutUint64(data[:], section)
511

512
	c.indexDb.Put(append([]byte("shead"), data[:]...), hash.Bytes())
513 514
}

515 516 517
// removeSectionHead removes the reference to a processed section from the index
// database.
func (c *ChainIndexer) removeSectionHead(section uint64) {
518
	var data [8]byte
519
	binary.BigEndian.PutUint64(data[:], section)
520 521 522

	c.indexDb.Delete(append([]byte("shead"), data[:]...))
}