~bzr-pqm/bzr/bzr.dev : contents of bzrlib/workingtree.py at revision 1157

~bzr-pqm/bzr/bzr.dev : (revision 1157)

# Copyright (C) 2005 Canonical Ltd

# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation; either version 2 of the License, or
# (at your option) any later version.

# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
# GNU General Public License for more details.

# You should have received a copy of the GNU General Public License
# along with this program; if not, write to the Free Software
# Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA  02111-1307  USA

# TODO: Don't allow WorkingTrees to be constructed for remote branches.

# FIXME: I don't know if writing out the cache from the destructor is really a
# good idea, because destructors are considered poor taste in Python, and
# it's not predictable when it will be written out.

import os
import fnmatch
        
import bzrlib.tree
from bzrlib.osutils import appendpath, file_kind, isdir, splitpath
from bzrlib.errors import BzrCheckError
from bzrlib.trace import mutter

class WorkingTree(bzrlib.tree.Tree):
    """Working copy tree.

    The inventory is held in the `Branch` working-inventory, and the
    files are in a directory on disk.

    It is possible for a `WorkingTree` to have a filename which is
    not listed in the Inventory and vice versa.
    """
    def __init__(self, basedir, inv):
        from bzrlib.hashcache import HashCache
        from bzrlib.trace import note, mutter

        self._inventory = inv
        self.basedir = basedir
        self.path2id = inv.path2id

        # update the whole cache up front and write to disk if anything changed;
        # in the future we might want to do this more selectively
        hc = self._hashcache = HashCache(basedir)
        hc.read()
        hc.scan()

        if hc.needs_write:
            mutter("write hc")
            hc.write()
            
            
    def __del__(self):
        if self._hashcache.needs_write:
            self._hashcache.write()


    def __iter__(self):
        """Iterate through file_ids for this tree.

        file_ids are in a WorkingTree if they are in the working inventory
        and the working file exists.
        """
        inv = self._inventory
        for path, ie in inv.iter_entries():
            if os.path.exists(self.abspath(path)):
                yield ie.file_id


    def __repr__(self):
        return "<%s of %s>" % (self.__class__.__name__,
                               getattr(self, 'basedir', None))



    def abspath(self, filename):
        return os.path.join(self.basedir, filename)

    def has_filename(self, filename):
        return os.path.exists(self.abspath(filename))

    def get_file(self, file_id):
        return self.get_file_byname(self.id2path(file_id))

    def get_file_byname(self, filename):
        return file(self.abspath(filename), 'rb')

    def _get_store_filename(self, file_id):
        ## XXX: badly named; this isn't in the store at all
        return self.abspath(self.id2path(file_id))

                
    def has_id(self, file_id):
        # files that have been deleted are excluded
        inv = self._inventory
        if not inv.has_id(file_id):
            return False
        path = inv.id2path(file_id)
        return os.path.exists(self.abspath(path))


    __contains__ = has_id
    

    def get_file_size(self, file_id):
        # is this still called?
        raise NotImplementedError()


    def get_file_sha1(self, file_id):
        path = self._inventory.id2path(file_id)
        return self._hashcache.get_sha1(path)


    def file_class(self, filename):
        if self.path2id(filename):
            return 'V'
        elif self.is_ignored(filename):
            return 'I'
        else:
            return '?'


    def list_files(self):
        """Recursively list all files as (path, class, kind, id).

        Lists, but does not descend into unversioned directories.

        This does not include files that have been deleted in this
        tree.

        Skips the control directory.
        """
        inv = self._inventory

        def descend(from_dir_relpath, from_dir_id, dp):
            ls = os.listdir(dp)
            ls.sort()
            for f in ls:
                ## TODO: If we find a subdirectory with its own .bzr
                ## directory, then that is a separate tree and we
                ## should exclude it.
                if bzrlib.BZRDIR == f:
                    continue

                # path within tree
                fp = appendpath(from_dir_relpath, f)

                # absolute path
                fap = appendpath(dp, f)
                
                f_ie = inv.get_child(from_dir_id, f)
                if f_ie:
                    c = 'V'
                elif self.is_ignored(fp):
                    c = 'I'
                else:
                    c = '?'

                fk = file_kind(fap)

                if f_ie:
                    if f_ie.kind != fk:
                        raise BzrCheckError("file %r entered as kind %r id %r, "
                                            "now of kind %r"
                                            % (fap, f_ie.kind, f_ie.file_id, fk))

                yield fp, c, fk, (f_ie and f_ie.file_id)

                if fk != 'directory':
                    continue

                if c != 'V':
                    # don't descend unversioned directories
                    continue
                
                for ff in descend(fp, f_ie.file_id, fap):
                    yield ff

        for f in descend('', inv.root.file_id, self.basedir):
            yield f
            


    def unknowns(self):
        for subp in self.extras():
            if not self.is_ignored(subp):
                yield subp


    def extras(self):
        """Yield all unknown files in this WorkingTree.

        If there are any unknown directories then only the directory is
        returned, not all its children.  But if there are unknown files
        under a versioned subdirectory, they are returned.

        Currently returned depth-first, sorted by name within directories.
        """
        ## TODO: Work from given directory downwards
        for path, dir_entry in self.inventory.directories():
            mutter("search for unknowns in %r" % path)
            dirabs = self.abspath(path)
            if not isdir(dirabs):
                # e.g. directory deleted
                continue

            fl = []
            for subf in os.listdir(dirabs):
                if (subf != '.bzr'
                    and (subf not in dir_entry.children)):
                    fl.append(subf)
            
            fl.sort()
            for subf in fl:
                subp = appendpath(path, subf)
                yield subp


    def ignored_files(self):
        """Yield list of PATH, IGNORE_PATTERN"""
        for subp in self.extras():
            pat = self.is_ignored(subp)
            if pat != None:
                yield subp, pat


    def get_ignore_list(self):
        """Return list of ignore patterns.

        Cached in the Tree object after the first call.
        """
        if hasattr(self, '_ignorelist'):
            return self._ignorelist

        l = bzrlib.DEFAULT_IGNORE[:]
        if self.has_filename(bzrlib.IGNORE_FILENAME):
            f = self.get_file_byname(bzrlib.IGNORE_FILENAME)
            l.extend([line.rstrip("\n\r") for line in f.readlines()])
        self._ignorelist = l
        return l


    def is_ignored(self, filename):
        r"""Check whether the filename matches an ignore pattern.

        Patterns containing '/' or '\' need to match the whole path;
        others match against only the last component.

        If the file is ignored, returns the pattern which caused it to
        be ignored, otherwise None.  So this can simply be used as a
        boolean if desired."""

        # TODO: Use '**' to match directories, and other extended
        # globbing stuff from cvs/rsync.

        # XXX: fnmatch is actually not quite what we want: it's only
        # approximately the same as real Unix fnmatch, and doesn't
        # treat dotfiles correctly and allows * to match /.
        # Eventually it should be replaced with something more
        # accurate.
        
        for pat in self.get_ignore_list():
            if '/' in pat or '\\' in pat:
                
                # as a special case, you can put ./ at the start of a
                # pattern; this is good to match in the top-level
                # only;
                
                if (pat[:2] == './') or (pat[:2] == '.\\'):
                    newpat = pat[2:]
                else:
                    newpat = pat
                if fnmatch.fnmatchcase(filename, newpat):
                    return pat
            else:
                if fnmatch.fnmatchcase(splitpath(filename)[-1], pat):
                    return pat
        else:
            return None
        

453 by Martin Pool - Split WorkingTree into its own file	1	# Copyright (C) 2005 Canonical Ltd
	2
	3	# This program is free software; you can redistribute it and/or modify
	4	# it under the terms of the GNU General Public License as published by
	5	# the Free Software Foundation; either version 2 of the License, or
	6	# (at your option) any later version.
	7
	8	# This program is distributed in the hope that it will be useful,
	9	# but WITHOUT ANY WARRANTY; without even the implied warranty of
	10	# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
	11	# GNU General Public License for more details.
	12
	13	# You should have received a copy of the GNU General Public License
	14	# along with this program; if not, write to the Free Software
	15	# Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
	16
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	17	# TODO: Don't allow WorkingTrees to be constructed for remote branches.
453 by Martin Pool - Split WorkingTree into its own file	18
956 by Martin Pool doc	19	# FIXME: I don't know if writing out the cache from the destructor is really a
	20	# good idea, because destructors are considered poor taste in Python, and
	21	# it's not predictable when it will be written out.
	22
453 by Martin Pool - Split WorkingTree into its own file	23	import os
1140 by Martin Pool - lift out import statements within WorkingTree	24	import fnmatch
	25
453 by Martin Pool - Split WorkingTree into its own file	26	import bzrlib.tree
1140 by Martin Pool - lift out import statements within WorkingTree	27	from bzrlib.osutils import appendpath, file_kind, isdir, splitpath
	28	from bzrlib.errors import BzrCheckError
	29	from bzrlib.trace import mutter
453 by Martin Pool - Split WorkingTree into its own file	30
	31	class WorkingTree(bzrlib.tree.Tree):
	32	"""Working copy tree.
	33
	34	The inventory is held in the `Branch` working-inventory, and the
	35	files are in a directory on disk.
	36
	37	It is possible for a `WorkingTree` to have a filename which is
	38	not listed in the Inventory and vice versa.
	39	"""
	40	def __init__(self, basedir, inv):
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	41	from bzrlib.hashcache import HashCache
	42	from bzrlib.trace import note, mutter
	43
453 by Martin Pool - Split WorkingTree into its own file	44	self._inventory = inv
	45	self.basedir = basedir
	46	self.path2id = inv.path2id
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	47
	48	# update the whole cache up front and write to disk if anything changed;
	49	# in the future we might want to do this more selectively
	50	hc = self._hashcache = HashCache(basedir)
	51	hc.read()
954 by Martin Pool - separate out code that just scans the hash cache to find files that are possibly	52	hc.scan()
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	53
	54	if hc.needs_write:
	55	mutter("write hc")
	56	hc.write()
954 by Martin Pool - separate out code that just scans the hash cache to find files that are possibly	57
	58
	59	def __del__(self):
	60	if self._hashcache.needs_write:
	61	self._hashcache.write()
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	62
453 by Martin Pool - Split WorkingTree into its own file	63
462 by Martin Pool - New form 'file_id in tree' to check if the file is present	64	def __iter__(self):
	65	"""Iterate through file_ids for this tree.
	66
	67	file_ids are in a WorkingTree if they are in the working inventory
	68	and the working file exists.
	69	"""
	70	inv = self._inventory
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	71	for path, ie in inv.iter_entries():
	72	if os.path.exists(self.abspath(path)):
	73	yield ie.file_id
462 by Martin Pool - New form 'file_id in tree' to check if the file is present	74
	75
453 by Martin Pool - Split WorkingTree into its own file	76	def __repr__(self):
453 by Martin Pool - Split WorkingTree into its own file	77	return "<%s of %s>" % (self.__class__.__name__,
954 by Martin Pool - separate out code that just scans the hash cache to find files that are possibly	78	getattr(self, 'basedir', None))
453 by Martin Pool - Split WorkingTree into its own file	79
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	80
	81
453 by Martin Pool - Split WorkingTree into its own file	82	def abspath(self, filename):
	83	return os.path.join(self.basedir, filename)
	84
	85	def has_filename(self, filename):
	86	return os.path.exists(self.abspath(filename))
	87
	88	def get_file(self, file_id):
	89	return self.get_file_byname(self.id2path(file_id))
	90
	91	def get_file_byname(self, filename):
	92	return file(self.abspath(filename), 'rb')
	93
	94	def _get_store_filename(self, file_id):
	95	## XXX: badly named; this isn't in the store at all
	96	return self.abspath(self.id2path(file_id))
	97
462 by Martin Pool - New form 'file_id in tree' to check if the file is present	98
453 by Martin Pool - Split WorkingTree into its own file	99	def has_id(self, file_id):
453 by Martin Pool - Split WorkingTree into its own file	100	# files that have been deleted are excluded
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	101	inv = self._inventory
	102	if not inv.has_id(file_id):
453 by Martin Pool - Split WorkingTree into its own file	103	return False
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	104	path = inv.id2path(file_id)
	105	return os.path.exists(self.abspath(path))
462 by Martin Pool - New form 'file_id in tree' to check if the file is present	106
	107
	108	__contains__ = has_id
	109
	110
453 by Martin Pool - Split WorkingTree into its own file	111	def get_file_size(self, file_id):
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	112	# is this still called?
	113	raise NotImplementedError()
453 by Martin Pool - Split WorkingTree into its own file	114
	115
	116	def get_file_sha1(self, file_id):
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	117	path = self._inventory.id2path(file_id)
	118	return self._hashcache.get_sha1(path)
453 by Martin Pool - Split WorkingTree into its own file	119
	120
	121	def file_class(self, filename):
	122	if self.path2id(filename):
	123	return 'V'
	124	elif self.is_ignored(filename):
	125	return 'I'
	126	else:
	127	return '?'
	128
	129
	130	def list_files(self):
	131	"""Recursively list all files as (path, class, kind, id).
	132
	133	Lists, but does not descend into unversioned directories.
	134
	135	This does not include files that have been deleted in this
	136	tree.
	137
	138	Skips the control directory.
	139	"""
866 by Martin Pool - use new path-based hashcache for WorkingTree- squash mtime/ctime to whole seconds- update and if necessary write out hashcache when WorkingTree object is created.	140	inv = self._inventory
453 by Martin Pool - Split WorkingTree into its own file	141
	142	def descend(from_dir_relpath, from_dir_id, dp):
	143	ls = os.listdir(dp)
	144	ls.sort()
	145	for f in ls:
	146	## TODO: If we find a subdirectory with its own .bzr
	147	## directory, then that is a separate tree and we
	148	## should exclude it.
	149	if bzrlib.BZRDIR == f:
	150	continue
	151
	152	# path within tree
	153	fp = appendpath(from_dir_relpath, f)
	154
	155	# absolute path
	156	fap = appendpath(dp, f)
	157
	158	f_ie = inv.get_child(from_dir_id, f)
	159	if f_ie:
	160	c = 'V'
	161	elif self.is_ignored(fp):
	162	c = 'I'
	163	else:
	164	c = '?'
	165
	166	fk = file_kind(fap)
	167
	168	if f_ie:
	169	if f_ie.kind != fk:
	170	raise BzrCheckError("file %r entered as kind %r id %r, "
	171	"now of kind %r"
	172	% (fap, f_ie.kind, f_ie.file_id, fk))
	173
	174	yield fp, c, fk, (f_ie and f_ie.file_id)
	175
	176	if fk != 'directory':
	177	continue
	178
	179	if c != 'V':
	180	# don't descend unversioned directories
	181	continue
	182
	183	for ff in descend(fp, f_ie.file_id, fap):
	184	yield ff
	185
	186	for f in descend('', inv.root.file_id, self.basedir):
	187	yield f
	188
	189
	190
	191	def unknowns(self):
	192	for subp in self.extras():
	193	if not self.is_ignored(subp):
	194	yield subp
	195
	196
	197	def extras(self):
	198	"""Yield all unknown files in this WorkingTree.
	199
	200	If there are any unknown directories then only the directory is
	201	returned, not all its children. But if there are unknown files
	202	under a versioned subdirectory, they are returned.
	203
	204	Currently returned depth-first, sorted by name within directories.
205	"""
206	## TODO: Work from given directory downwards
207	for path, dir_entry in self.inventory.directories():
208	mutter("search for unknowns in %r" % path)
209	dirabs = self.abspath(path)
210	if not isdir(dirabs):
211	# e.g. directory deleted
212	continue
213
214	fl = []
215	for subf in os.listdir(dirabs):
216	if (subf != '.bzr'
217	and (subf not in dir_entry.children)):
218	fl.append(subf)
219
220	fl.sort()
221	for subf in fl:
222	subp = appendpath(path, subf)
223	yield subp
224
225
226	def ignored_files(self):
227	"""Yield list of PATH, IGNORE_PATTERN"""
228	for subp in self.extras():
229	pat = self.is_ignored(subp)
230	if pat != None:
231	yield subp, pat
232
233
234	def get_ignore_list(self):
235	"""Return list of ignore patterns.
236
237	Cached in the Tree object after the first call.
238	"""
239	if hasattr(self, '_ignorelist'):
240	return self._ignorelist
241
242	l = bzrlib.DEFAULT_IGNORE[:]
243	if self.has_filename(bzrlib.IGNORE_FILENAME):
244	f = self.get_file_byname(bzrlib.IGNORE_FILENAME)
245	l.extend([line.rstrip("\n\r") for line in f.readlines()])
246	self._ignorelist = l
247	return l
248
249
250	def is_ignored(self, filename):
251	r"""Check whether the filename matches an ignore pattern.
252
253	Patterns containing '/' or '\' need to match the whole path;
254	others match against only the last component.
255
256	If the file is ignored, returns the pattern which caused it to
257	be ignored, otherwise None. So this can simply be used as a
258	boolean if desired."""
259
260	# TODO: Use '**' to match directories, and other extended
261	# globbing stuff from cvs/rsync.
262
263	# XXX: fnmatch is actually not quite what we want: it's only
264	# approximately the same as real Unix fnmatch, and doesn't
265	# treat dotfiles correctly and allows * to match /.
266	# Eventually it should be replaced with something more
267	# accurate.
268
269	for pat in self.get_ignore_list():
270	if '/' in pat or '\\' in pat:
271
272	# as a special case, you can put ./ at the start of a
273	# pattern; this is good to match in the top-level
274	# only;
275
276	if (pat[:2] == './') or (pat[:2] == '.\\'):
277	newpat = pat[2:]
278	else:
279	newpat = pat
280	if fnmatch.fnmatchcase(filename, newpat):
281	return pat
282	else:
283	if fnmatch.fnmatchcase(splitpath(filename)[-1], pat):
284	return pat
285	else:
286	return None
1140 by Martin Pool - lift out import statements within WorkingTree	287